agora inbox for [email protected]help / color / mirror / Atom feed
[PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. 266+ messages / 2 participants [nested] [flat]
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. @ 2020-06-12 02:38 Andres Freund <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Andres Freund @ 2020-06-12 02:38 UTC (permalink / raw) --- src/include/portability/instr_time.h | 68 ++++++++++++++++++++++++---- 1 file changed, 60 insertions(+), 8 deletions(-) diff --git a/src/include/portability/instr_time.h b/src/include/portability/instr_time.h index fc058d548a8..8b2f9a2e707 100644 --- a/src/include/portability/instr_time.h +++ b/src/include/portability/instr_time.h @@ -83,7 +83,9 @@ #define PG_INSTR_CLOCK CLOCK_REALTIME #endif +/* time in baseline cpu cycles */ typedef int64 instr_time; + #define NS_PER_S INT64CONST(1000000000) #define US_PER_S INT64CONST(1000000) #define MS_PER_S INT64CONST(1000) @@ -95,17 +97,67 @@ typedef int64 instr_time; #define INSTR_TIME_SET_ZERO(t) ((t) = 0) -static inline instr_time pg_clock_gettime_ns(void) +#include <x86intrin.h> +#include <cpuid.h> + +/* + * Return what the number of cycles needs to be multiplied with to end up with + * seconds. + * + * FIXME: The cold portion should probably be out-of-line. And it'd be better + * to not recompute this in every file that uses this. Best would probably be + * to require explicit initialization of cycles_to_sec, because having a + * branch really is unnecessary. + * + * FIXME: We should probably not unnecessarily use floating point math + * here. And it's likely that the numbers are small enough that we are running + * into floating point inaccuracies already. Probably worthwhile to be a good + * bit smarter. + * + * FIXME: This would need to be conditional, with a fallback to something not + * rdtsc based. + */ +static inline double __attribute__((const)) +get_cycles_to_sec(void) { - struct timespec tmp; + static double cycles_to_sec = 0; - clock_gettime(PG_INSTR_CLOCK, &tmp); + /* + * Compute baseline cpu peformance, determines speed at which rdtsc advances + */ + if (unlikely(cycles_to_sec == 0)) + { + uint32 cpuinfo[4] = {0}; - return tmp.tv_sec * NS_PER_S + tmp.tv_nsec; + __get_cpuid(0x16, cpuinfo, cpuinfo + 1, cpuinfo + 2, cpuinfo + 3); + cycles_to_sec = 1 / ((double) cpuinfo[0] * 1000 * 1000); + } + + return cycles_to_sec; +} + +static inline instr_time pg_clock_gettime_ref_cycles(void) +{ + /* + * The rdtscp waits for all in-flight instructions to finish (but allows + * later instructions to start concurrently). That's good for some timing + * situations (when the time is supposed to cover all the work), but + * terrible for others (when sub-parts of work are measured, because then + * the pipeline stall due to the wait change the overall timing). + */ +#if 0 + unsigned int aux; + int64 tsc = __rdtscp(&aux); + + return tsc; +#else + + return __rdtsc(); +#endif } #define INSTR_TIME_SET_CURRENT(t) \ - (t) = pg_clock_gettime_ns() + (t) = pg_clock_gettime_ref_cycles() #define INSTR_TIME_ADD(x,y) \ do { \ @@ -123,13 +175,13 @@ static inline instr_time pg_clock_gettime_ns(void) } while (0) #define INSTR_TIME_GET_DOUBLE(t) \ - ((double) (t) / NS_PER_S) + ((double) (t) * get_cycles_to_sec()) #define INSTR_TIME_GET_MILLISEC(t) \ - ((double) (t) / NS_PER_MS) + ((double) (t) * (get_cycles_to_sec() * MS_PER_S)) #define INSTR_TIME_GET_MICROSEC(t) \ - ((double) (t) / NS_PER_US) + ((double) (t) * (get_cycles_to_sec() * US_PER_S)) #else /* !HAVE_CLOCK_GETTIME */ -- 2.25.0.114.g5b0ca878e0 --wn4ncs637ccpvbhb-- ^ permalink raw reply [nested|flat] 266+ messages in thread
* [PATCH v22 6/8] Row pattern recognition patch (docs). @ 2024-09-19 04:48 Tatsuo Ishii <[email protected]> 0 siblings, 0 replies; 266+ messages in thread From: Tatsuo Ishii @ 2024-09-19 04:48 UTC (permalink / raw) --- doc/src/sgml/advanced.sgml | 82 ++++++++++++++++++++++++++++++++++++ doc/src/sgml/func.sgml | 54 ++++++++++++++++++++++++ doc/src/sgml/ref/select.sgml | 38 ++++++++++++++++- 3 files changed, 172 insertions(+), 2 deletions(-) diff --git a/doc/src/sgml/advanced.sgml b/doc/src/sgml/advanced.sgml index 755c9f1485..b0b1d1c51e 100644 --- a/doc/src/sgml/advanced.sgml +++ b/doc/src/sgml/advanced.sgml @@ -537,6 +537,88 @@ WHERE pos < 3; <literal>rank</literal> less than 3. </para> + <para> + Row pattern common syntax can be used to perform row pattern recognition + in a query. The row pattern common syntax includes two sub + clauses: <literal>DEFINE</literal> + and <literal>PATTERN</literal>. <literal>DEFINE</literal> defines + definition variables along with an expression. The expression must be a + logical expression, which means it must + return <literal>TRUE</literal>, <literal>FALSE</literal> + or <literal>NULL</literal>. The expression may comprise column references + and functions. Window functions, aggregate functions and subqueries are + not allowed. An example of <literal>DEFINE</literal> is as follows. + +<programlisting> +DEFINE + LOWPRICE AS price <= 100, + UP AS price > PREV(price), + DOWN AS price < PREV(price) +</programlisting> + + Note that <function>PREV</function> returns the price column in the + previous row if it's called in a context of row pattern recognition. Thus in + the second line the definition variable "UP" is <literal>TRUE</literal> + when the price column in the current row is greater than the price column + in the previous row. Likewise, "DOWN" is <literal>TRUE</literal> when when + the price column in the current row is lower than the price column in the + previous row. + </para> + <para> + Once <literal>DEFINE</literal> exists, <literal>PATTERN</literal> can be + used. <literal>PATTERN</literal> defines a sequence of rows that satisfies + certain conditions. For example following <literal>PATTERN</literal> + defines that a row starts with the condition "LOWPRICE", then one or more + rows satisfy "UP" and finally one or more rows satisfy "DOWN". Note that + "+" means one or more matches. Also you can use "*", which means zero or + more matches. If a sequence of rows which satisfies the PATTERN is found, + in the starting row of the sequence of rows all window functions and + aggregates are shown in the target list. Note that aggregations only look + into the matched rows, rather than whole frame. On the second or + subsequent rows all window functions are NULL. Aggregates are NULL or 0 + (count case) depending on its aggregation definition. For rows that do not + match on the PATTERN, all window functions and aggregates are shown AS + NULL too, except count showing 0. This is because the rows do not match, + thus they are in an empty frame. Example of a <literal>SELECT</literal> + using the <literal>DEFINE</literal> and <literal>PATTERN</literal> clause + is as follows. + +<programlisting> +SELECT company, tdate, price, + first_value(price) OVER w, + max(price) OVER w, + count(price) OVER w +FROM stock + WINDOW w AS ( + PARTITION BY company + ORDER BY tdate + ROWS BETWEEN CURRENT ROW AND UNBOUNDED FOLLOWING + AFTER MATCH SKIP PAST LAST ROW + INITIAL + PATTERN (LOWPRICE UP+ DOWN+) + DEFINE + LOWPRICE AS price <= 100, + UP AS price > PREV(price), + DOWN AS price < PREV(price) +); +</programlisting> +<screen> + company | tdate | price | first_value | max | count +----------+------------+-------+-------------+-----+------- + company1 | 2023-07-01 | 100 | 100 | 200 | 4 + company1 | 2023-07-02 | 200 | | | 0 + company1 | 2023-07-03 | 150 | | | 0 + company1 | 2023-07-04 | 140 | | | 0 + company1 | 2023-07-05 | 150 | | | 0 + company1 | 2023-07-06 | 90 | 90 | 130 | 4 + company1 | 2023-07-07 | 110 | | | 0 + company1 | 2023-07-08 | 130 | | | 0 + company1 | 2023-07-09 | 120 | | | 0 + company1 | 2023-07-10 | 130 | | | 0 +(10 rows) +</screen> + </para> + <para> When a query involves multiple window functions, it is possible to write out each one with a separate <literal>OVER</literal> clause, but this is diff --git a/doc/src/sgml/func.sgml b/doc/src/sgml/func.sgml index 6f75bd0c7d..4482c80a70 100644 --- a/doc/src/sgml/func.sgml +++ b/doc/src/sgml/func.sgml @@ -23261,6 +23261,7 @@ SELECT count(*) FROM sometable; returns <literal>NULL</literal> if there is no such row. </para></entry> </row> + </tbody> </tgroup> </table> @@ -23300,6 +23301,59 @@ SELECT count(*) FROM sometable; Other frame specifications can be used to obtain other effects. </para> + <para> + Row pattern recognition navigation functions are listed in + <xref linkend="functions-rpr-navigation-table"/>. These functions + can be used to describe DEFINE clause of Row pattern recognition. + </para> + + <table id="functions-rpr-navigation-table"> + <title>Row Pattern Navigation Functions</title> + <tgroup cols="1"> + <thead> + <row> + <entry role="func_table_entry"><para role="func_signature"> + Function + </para> + <para> + Description + </para></entry> + </row> + </thead> + + <tbody> + <row> + <entry role="func_table_entry"><para role="func_signature"> + <indexterm> + <primary>prev</primary> + </indexterm> + <function>prev</function> ( <parameter>value</parameter> <type>anyelement</type> ) + <returnvalue>anyelement</returnvalue> + </para> + <para> + Returns the column value at the previous row; + returns NULL if there is no previous row in the window frame. + </para></entry> + </row> + + <row> + <entry role="func_table_entry"><para role="func_signature"> + <indexterm> + <primary>next</primary> + </indexterm> + <function>next</function> ( <parameter>value</parameter> <type>anyelement</type> ) + <returnvalue>anyelement</returnvalue> + </para> + <para> + Returns the column value at the next row; + returns NULL if there is no next row in the window frame. + </para></entry> + </row> + + </tbody> + </tgroup> + </table> + <note> <para> The SQL standard defines a <literal>RESPECT NULLS</literal> or diff --git a/doc/src/sgml/ref/select.sgml b/doc/src/sgml/ref/select.sgml index d7089eac0b..7e1c9989ba 100644 --- a/doc/src/sgml/ref/select.sgml +++ b/doc/src/sgml/ref/select.sgml @@ -969,8 +969,8 @@ WINDOW <replaceable class="parameter">window_name</replaceable> AS ( <replaceabl The <replaceable class="parameter">frame_clause</replaceable> can be one of <synopsis> -{ RANGE | ROWS | GROUPS } <replaceable>frame_start</replaceable> [ <replaceable>frame_exclusion</replaceable> ] -{ RANGE | ROWS | GROUPS } BETWEEN <replaceable>frame_start</replaceable> AND <replaceable>frame_end</replaceable> [ <replaceable>frame_exclusion</replaceable> ] +{ RANGE | ROWS | GROUPS } <replaceable>frame_start</replaceable> [ <replaceable>frame_exclusion</replaceable> ] [row_pattern_common_syntax] +{ RANGE | ROWS | GROUPS } BETWEEN <replaceable>frame_start</replaceable> AND <replaceable>frame_end</replaceable> [ <replaceable>frame_exclusion</replaceable> ] [row_pattern_common_syntax] </synopsis> where <replaceable>frame_start</replaceable> @@ -1077,6 +1077,40 @@ EXCLUDE NO OTHERS a given peer group will be in the frame or excluded from it. </para> + <para> + The + optional <replaceable class="parameter">row_pattern_common_syntax</replaceable> + defines the <firstterm>row pattern recognition condition</firstterm> for + this + window. <replaceable class="parameter">row_pattern_common_syntax</replaceable> + includes following subclauses. <literal>AFTER MATCH SKIP PAST LAST + ROW</literal> or <literal>AFTER MATCH SKIP TO NEXT ROW</literal> controls + how to proceed to next row position after a match + found. With <literal>AFTER MATCH SKIP PAST LAST ROW</literal> (the + default) next row position is next to the last row of previous match. On + the other hand, with <literal>AFTER MATCH SKIP TO NEXT ROW</literal> next + row position is always next to the last row of previous + match. <literal>DEFINE</literal> defines definition variables along with a + boolean expression. <literal>PATTERN</literal> defines a sequence of rows + that satisfies certain conditions using variables defined + in <literal>DEFINE</literal> clause. If the variable is not defined in + the <literal>DEFINE</literal> clause, it is implicitly assumed + following is defined in the <literal>DEFINE</literal> clause. + +<synopsis> +<literal>variable_name</literal> AS TRUE +</synopsis> + + Note that the maximu number of variables defined + in <literal>DEFINE</literal> clause is 26. + +<synopsis> +[ AFTER MATCH SKIP PAST LAST ROW | AFTER MATCH SKIP TO NEXT ROW ] +PATTERN <replaceable class="parameter">pattern_variable_name</replaceable>[+] [, ...] +DEFINE <replaceable class="parameter">definition_varible_name</replaceable> AS <replaceable class="parameter">expression</replaceable> [, ...] +</synopsis> + </para> + <para> The purpose of a <literal>WINDOW</literal> clause is to specify the behavior of <firstterm>window functions</firstterm> appearing in the query's -- 2.25.1 ----Next_Part(Thu_Sep_19_13_59_47_2024_608)-- Content-Type: Text/X-Patch; charset=us-ascii Content-Transfer-Encoding: 7bit Content-Disposition: inline; filename="v22-0007-Row-pattern-recognition-patch-tests.patch" ^ permalink raw reply [nested|flat] 266+ messages in thread
end of thread, other threads:[~2024-09-19 04:48 UTC | newest] Thread overview: 266+ messages (download: mbox mbox.gz follow: Atom feed) -- links below jump to the message on this page -- 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2020-06-12 02:38 [PATCH v1 2/2] WIP: Use cpu reference cycles, via rdtsc, to measure time for instrumentation. Andres Freund <[email protected]> 2024-09-19 04:48 [PATCH v22 6/8] Row pattern recognition patch (docs). Tatsuo Ishii <[email protected]>
This inbox is served by agora; see mirroring instructions for how to clone and mirror all data and code used for this inbox