agora inbox for [email protected]help / color / mirror / Atom feed
[PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. 266+ messages / 2 participants [nested] [flat]
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 6/6] Handle pg_get_triggerdef default args in system_functions.sql @ 2025-12-09 19:51 Mark Wong <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Mark Wong @ 2025-12-09 19:51 UTC (permalink / raw) Modernize pg_get_triggerdef to use CREATE OR REPLACE FUNCTION to handle the optional pretty argument. --- src/backend/catalog/system_functions.sql | 7 +++++++ src/backend/utils/adt/ruleutils.c | 14 -------------- src/include/catalog/pg_proc.dat | 5 +---- 3 files changed, 8 insertions(+), 18 deletions(-) diff --git a/src/backend/catalog/system_functions.sql b/src/backend/catalog/system_functions.sql index 81d210a9c45..c9c011ea5c1 100644 --- a/src/backend/catalog/system_functions.sql +++ b/src/backend/catalog/system_functions.sql @@ -699,6 +699,13 @@ LANGUAGE INTERNAL PARALLEL SAFE AS 'pg_get_expr'; +CREATE OR REPLACE FUNCTION + pg_get_triggerdef(trigger oid, pretty bool DEFAULT false) +RETURNS TEXT +LANGUAGE INTERNAL +PARALLEL SAFE +AS 'pg_get_triggerdef'; + -- -- The default permissions for functions mean that anyone can execute them. -- A number of functions shouldn't be executable by just anyone, but rather diff --git a/src/backend/utils/adt/ruleutils.c b/src/backend/utils/adt/ruleutils.c index 9effa02fa2e..45a35f823e8 100644 --- a/src/backend/utils/adt/ruleutils.c +++ b/src/backend/utils/adt/ruleutils.c @@ -807,20 +807,6 @@ pg_get_viewdef_worker(Oid viewoid, int prettyFlags, int wrapColumn) */ Datum pg_get_triggerdef(PG_FUNCTION_ARGS) -{ - Oid trigid = PG_GETARG_OID(0); - char *res; - - res = pg_get_triggerdef_worker(trigid, false); - - if (res == NULL) - PG_RETURN_NULL(); - - PG_RETURN_TEXT_P(string_to_text(res)); -} - -Datum -pg_get_triggerdef_ext(PG_FUNCTION_ARGS) { Oid trigid = PG_GETARG_OID(0); bool pretty = PG_GETARG_BOOL(1); diff --git a/src/include/catalog/pg_proc.dat b/src/include/catalog/pg_proc.dat index 10c5286bd7d..146963c254b 100644 --- a/src/include/catalog/pg_proc.dat +++ b/src/include/catalog/pg_proc.dat @@ -3974,9 +3974,6 @@ proname => 'pg_get_partition_constraintdef', provolatile => 's', prorettype => 'text', proargtypes => 'oid', prosrc => 'pg_get_partition_constraintdef' }, -{ oid => '1662', descr => 'trigger description', - proname => 'pg_get_triggerdef', provolatile => 's', prorettype => 'text', - proargtypes => 'oid', prosrc => 'pg_get_triggerdef' }, { oid => '1665', descr => 'name of sequence for a serial column', proname => 'pg_get_serial_sequence', provolatile => 's', prorettype => 'text', proargtypes => 'text text', prosrc => 'pg_get_serial_sequence' }, @@ -8535,7 +8532,7 @@ prosrc => 'pg_timezone_names' }, { oid => '2730', descr => 'trigger description with pretty-print option', proname => 'pg_get_triggerdef', provolatile => 's', prorettype => 'text', - proargtypes => 'oid bool', prosrc => 'pg_get_triggerdef_ext' }, + proargtypes => 'oid bool', prosrc => 'pg_get_triggerdef' }, # asynchronous notifications { oid => '3035', -- 2.51.2 --IRGsBYp1UN8d6WrT-- ^ permalink raw reply [nested|flat] 266+ messages in thread
end of thread, other threads:[~2025-12-09 19:51 UTC | newest] Thread overview: 266+ messages (download: mbox mbox.gz follow: Atom feed) -- links below jump to the message on this page -- 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2025-12-09 19:51 [PATCH v1 6/6] Handle pg_get_triggerdef default args in system_functions.sql Mark Wong <[email protected]>
This inbox is served by agora; see mirroring instructions for how to clone and mirror all data and code used for this inbox