The "competition tests" are being made portable (#907)

* More portable competition

* This will enable SIMDJSON_COMPETITION everywhere by default.

* Minor fixes
This commit is contained in:
Daniel Lemire
2020-05-31 20:34:06 -04:00
committed by GitHub
parent f772bf4fbc
commit 2fe2dd170b
5 changed files with 93 additions and 207 deletions
+62 -183
View File
@@ -1,82 +1,7 @@
#ifndef _BENCHMARK_H_
#define _BENCHMARK_H_
#include <float.h>
#include <stdint.h>
#include <time.h>
#ifdef __x86_64__
const char *unitname = "cycles";
#define RDTSC_START(cycles) \
do { \
uint32_t cyc_high, cyc_low; \
__asm volatile("cpuid\n" \
"rdtsc\n" \
"mov %%edx, %0\n" \
"mov %%eax, %1" \
: "=r"(cyc_high), "=r"(cyc_low) \
: \
: /* no read only */ \
"%rax", "%rbx", "%rcx", "%rdx" /* clobbers */ \
); \
(cycles) = ((uint64_t)cyc_high << 32) | cyc_low; \
} while (0)
#define RDTSC_STOP(cycles) \
do { \
uint32_t cyc_high, cyc_low; \
__asm volatile("rdtscp\n" \
"mov %%edx, %0\n" \
"mov %%eax, %1\n" \
"cpuid" \
: "=r"(cyc_high), "=r"(cyc_low) \
: /* no read only registers */ \
: "%rax", "%rbx", "%rcx", "%rdx" /* clobbers */ \
); \
(cycles) = ((uint64_t)cyc_high << 32) | cyc_low; \
} while (0)
#else
const char *unitname = " (clock units) ";
#define RDTSC_START(cycles) \
do { \
cycles = clock(); \
} while (0)
#define RDTSC_STOP(cycles) \
do { \
cycles = clock(); \
} while (0)
#endif
static __attribute__((noinline)) uint64_t rdtsc_overhead_func(uint64_t dummy) {
return dummy;
}
uint64_t global_rdtsc_overhead = (uint64_t)UINT64_MAX;
#define RDTSC_SET_OVERHEAD(test, repeat) \
do { \
uint64_t cycles_start, cycles_final, cycles_diff; \
uint64_t min_diff = UINT64_MAX; \
for (decltype(repeat) i = 0; i < repeat; i++) { \
__asm volatile("" ::: /* pretend to clobber */ "memory"); \
RDTSC_START(cycles_start); \
test; \
RDTSC_STOP(cycles_final); \
cycles_diff = (cycles_final - cycles_start); \
if (cycles_diff < min_diff) \
min_diff = cycles_diff; \
} \
global_rdtsc_overhead = min_diff; \
} while (0)
double diff(timespec start, timespec end) {
return static_cast<double>((end.tv_nsec + 1000000000 * end.tv_sec) -
(start.tv_nsec + 1000000000 * start.tv_sec)) /
1000000000.0;
}
#include "event_counter.h"
/*
* Prints the best number of operations per cycle where
@@ -86,62 +11,48 @@ double diff(timespec start, timespec end) {
*/
#define BEST_TIME(name, test, expected, pre, repeat, size, verbose) \
do { \
if (global_rdtsc_overhead == UINT64_MAX) { \
RDTSC_SET_OVERHEAD(rdtsc_overhead_func(1), repeat); \
} \
if (verbose) \
printf("%-40s\t: ", name); \
else \
printf("\"%-40s\"", name); \
fflush(NULL); \
uint64_t cycles_start, cycles_final, cycles_diff; \
uint64_t min_diff = (uint64_t)-1; \
double min_sumclockdiff = DBL_MAX; \
uint64_t sum_diff = 0; \
double sumclockdiff = 0; \
struct timespec time1, time2; \
for (decltype(repeat) i = 0; i < repeat; i++) { \
event_collector collector; \
event_aggregate aggregate{}; \
for (decltype(repeat) i = 0; i < repeat; i++) { \
pre; \
__asm volatile("" ::: /* pretend to clobber */ "memory"); \
clock_gettime(CLOCK_REALTIME, &time1); \
RDTSC_START(cycles_start); \
std::atomic_thread_fence(std::memory_order_acquire); \
collector.start(); \
if (test != expected) { \
fprintf(stderr, "not expected (%d , %d )", (int)test, (int)expected); \
break; \
} \
RDTSC_STOP(cycles_final); \
clock_gettime(CLOCK_REALTIME, &time2); \
double thistiming = diff(time1, time2); \
sumclockdiff += thistiming; \
if (thistiming < min_sumclockdiff) \
min_sumclockdiff = thistiming; \
cycles_diff = (cycles_final - cycles_start - global_rdtsc_overhead); \
if (cycles_diff < min_diff) \
min_diff = cycles_diff; \
sum_diff += cycles_diff; \
std::atomic_thread_fence(std::memory_order_release); \
event_count allocate_count = collector.end(); \
aggregate << allocate_count; \
} \
uint64_t S = size; \
double cycle_per_op = static_cast<double>(min_diff) / static_cast<double>(S); \
double avg_cycle_per_op = static_cast<double>(sum_diff) / (static_cast<double>(S) * static_cast<double>(repeat)); \
double avg_gb_per_s = \
(static_cast<double>(S) * static_cast<double>(repeat)) / ((sumclockdiff)*1000.0 * 1000.0 * 1000.0); \
double max_gb_per_s = \
static_cast<double>(S) / (min_sumclockdiff * 1000.0 * 1000.0 * 1000.0); \
if (verbose) \
printf(" %7.3f %s per input byte (best) ", cycle_per_op, unitname); \
if (verbose) \
printf(" %7.3f %s (avg) ", avg_cycle_per_op, unitname); \
if (verbose) \
printf(" %7.3f GB/s (error margin: %5.3f GB/s)", max_gb_per_s, \
-avg_gb_per_s + max_gb_per_s); \
if (verbose) \
printf(" %13.0f documents/s (best)", 1.0/min_sumclockdiff); \
if (verbose) \
printf(" %13.0f documents/s (avg)", 1.0/(sumclockdiff/static_cast<double>(repeat))); \
if (!verbose) \
printf(" %20.3f %20.3f %20.3f %20.3f", cycle_per_op, \
avg_cycle_per_op - cycle_per_op, max_gb_per_s, \
-avg_gb_per_s + max_gb_per_s); \
if (collector.has_events()) { \
printf("%7.3f", aggregate.best.cycles() / static_cast<double>(size)); \
if (verbose) { \
printf(" cycles/byte "); \
} \
printf("\t"); \
printf("%7.3f", \
aggregate.best.instructions() / static_cast<double>(size)); \
if (verbose) { \
printf(" instructions/byte "); \
} \
printf("\t"); \
} \
double gb = static_cast<double>(size) / 1000000000.0; \
printf("%7.3f", gb / aggregate.best.elapsed_sec()); \
if (verbose) { \
printf(" GB/s "); \
} \
printf("%7.3f", 1.0 / aggregate.best.elapsed_sec()); \
if (verbose) { \
printf(" documents/s "); \
} \
printf("\n"); \
fflush(NULL); \
} while (0)
@@ -149,78 +60,46 @@ double diff(timespec start, timespec end) {
// like BEST_TIME, but no check
#define BEST_TIME_NOCHECK(name, test, pre, repeat, size, verbose) \
do { \
if (global_rdtsc_overhead == UINT64_MAX) { \
RDTSC_SET_OVERHEAD(rdtsc_overhead_func(1), repeat); \
} \
if (verbose) \
printf("%-40s\t: ", name); \
else \
printf("\"%-40s\"", name); \
fflush(NULL); \
uint64_t cycles_start, cycles_final, cycles_diff; \
uint64_t min_diff = (uint64_t)-1; \
uint64_t sum_diff = 0; \
for (int i = 0; i < repeat; i++) { \
event_collector collector; \
event_aggregate aggregate{}; \
for (decltype(repeat) i = 0; i < repeat; i++) { \
pre; \
__asm volatile("" ::: /* pretend to clobber */ "memory"); \
RDTSC_START(cycles_start); \
std::atomic_thread_fence(std::memory_order_acquire); \
collector.start(); \
test; \
RDTSC_STOP(cycles_final); \
cycles_diff = (cycles_final - cycles_start - global_rdtsc_overhead); \
if (cycles_diff < min_diff) \
min_diff = cycles_diff; \
sum_diff += cycles_diff; \
std::atomic_thread_fence(std::memory_order_release); \
event_count allocate_count = collector.end(); \
aggregate << allocate_count; \
} \
uint64_t S = size; \
double cycle_per_op = static_cast<double>(min_diff) / static_cast<double>(S); \
double avg_cycle_per_op = static_cast<double>(sum_diff) / (static_cast<double>(S) * static_cast<double>(repeat)); \
if (verbose) \
printf(" %.3f %s per input byte (best) ", cycle_per_op, unitname); \
if (verbose) \
printf(" %.3f %s per input byte (avg) ", avg_cycle_per_op, unitname); \
if (verbose) \
printf("\n"); \
if (!verbose) \
printf(" %.3f ", cycle_per_op); \
fflush(NULL); \
} while (0)
// like BEST_TIME except that we run a function to check the result
#define BEST_TIME_CHECK(test, check, pre, repeat, size, verbose) \
do { \
if (global_rdtsc_overhead == UINT64_MAX) { \
RDTSC_SET_OVERHEAD(rdtsc_overhead_func(1), repeat); \
} \
if (verbose) \
printf("%-60s\t:\n", #test); \
fflush(NULL); \
uint64_t cycles_start, cycles_final, cycles_diff; \
uint64_t min_diff = (uint64_t)-1; \
uint64_t sum_diff = 0; \
for (int i = 0; i < repeat; i++) { \
pre; \
__asm volatile("" ::: /* pretend to clobber */ "memory"); \
RDTSC_START(cycles_start); \
test; \
RDTSC_STOP(cycles_final); \
if (!check) { \
printf("error"); \
break; \
if (collector.has_events()) { \
printf("%7.3f", aggregate.best.cycles() / static_cast<double>(size)); \
if (verbose) { \
printf(" cycles/byte "); \
} \
cycles_diff = (cycles_final - cycles_start - global_rdtsc_overhead); \
if (cycles_diff < min_diff) \
min_diff = cycles_diff; \
sum_diff += cycles_diff; \
printf("\t"); \
printf("%7.3f", \
aggregate.best.instructions() / static_cast<double>(size)); \
if (verbose) { \
printf(" instructions/byte "); \
} \
printf("\t"); \
} \
uint64_t S = size; \
float cycle_per_op = (min_diff) / (double)S; \
float avg_cycle_per_op = (sum_diff) / ((double)S * repeat); \
if (verbose) \
printf(" %.3f cycles per operation (best) ", cycle_per_op); \
if (verbose) \
printf("\t%.3f cycles per operation (avg) ", avg_cycle_per_op); \
if (verbose) \
printf("\n"); \
if (!verbose) \
printf(" %.3f ", cycle_per_op); \
double gb = static_cast<double>(size) / 1000000000.0; \
printf("%7.3f", gb / aggregate.best.elapsed_sec()); \
if (verbose) { \
printf(" GB/s "); \
} \
printf("%7.3f", 1.0 / aggregate.best.elapsed_sec()); \
if (verbose) { \
printf(" documents/s "); \
} \
printf("\n"); \
fflush(NULL); \
} while (0)
+9 -9
View File
@@ -107,7 +107,7 @@ void simdjson_recurse(std::vector<int64_t> & v, simdjson::dom::element element)
}
}
__attribute__((noinline)) std::vector<int64_t>
really_inline std::vector<int64_t>
simdjson_just_dom(simdjson::dom::element doc) {
std::vector<int64_t> answer;
simdjson_recurse(answer, doc);
@@ -115,7 +115,7 @@ simdjson_just_dom(simdjson::dom::element doc) {
return answer;
}
__attribute__((noinline)) std::vector<int64_t>
really_inline std::vector<int64_t>
simdjson_compute_stats(const simdjson::padded_string &p) {
std::vector<int64_t> answer;
simdjson::dom::parser parser;
@@ -129,7 +129,7 @@ simdjson_compute_stats(const simdjson::padded_string &p) {
return answer;
}
__attribute__((noinline)) simdjson::error_code
really_inline simdjson::error_code
simdjson_just_parse(const simdjson::padded_string &p) {
simdjson::dom::parser parser;
return parser.parse(p).error();
@@ -187,7 +187,7 @@ void sajson_traverse(std::vector<int64_t> &answer, const sajson::value &node) {
}
}
__attribute__((noinline)) std::vector<int64_t>
really_inline std::vector<int64_t>
sasjon_just_dom(sajson::document &d) {
std::vector<int64_t> answer;
sajson_traverse(answer, d.get_root());
@@ -195,7 +195,7 @@ sasjon_just_dom(sajson::document &d) {
return answer;
}
__attribute__((noinline)) std::vector<int64_t>
really_inline std::vector<int64_t>
sasjon_compute_stats(const simdjson::padded_string &p) {
std::vector<int64_t> answer;
char *buffer = (char *)malloc(p.size());
@@ -212,7 +212,7 @@ sasjon_compute_stats(const simdjson::padded_string &p) {
return answer;
}
__attribute__((noinline)) bool
really_inline bool
sasjon_just_parse(const simdjson::padded_string &p) {
char *buffer = (char *)malloc(p.size());
memcpy(buffer, p.data(), p.size());
@@ -263,7 +263,7 @@ void rapid_traverse(std::vector<int64_t> &answer, const rapidjson::Value &v) {
}
}
__attribute__((noinline)) std::vector<int64_t>
really_inline std::vector<int64_t>
rapid_just_dom(rapidjson::Document &d) {
std::vector<int64_t> answer;
rapid_traverse(answer, d);
@@ -271,7 +271,7 @@ rapid_just_dom(rapidjson::Document &d) {
return answer;
}
__attribute__((noinline)) std::vector<int64_t>
really_inline std::vector<int64_t>
rapid_compute_stats(const simdjson::padded_string &p) {
std::vector<int64_t> answer;
char *buffer = (char *)malloc(p.size() + 1);
@@ -289,7 +289,7 @@ rapid_compute_stats(const simdjson::padded_string &p) {
return answer;
}
__attribute__((noinline)) bool
really_inline bool
rapid_just_parse(const simdjson::padded_string &p) {
char *buffer = (char *)malloc(p.size() + 1);
memcpy(buffer, p.data(), p.size());
+7 -7
View File
@@ -95,7 +95,7 @@ void simdjson_recurse(stat_t &s, simdjson::dom::element element) {
}
}
__attribute__((noinline)) stat_t
never_inline stat_t
simdjson_compute_stats(const simdjson::padded_string &p) {
stat_t s{};
simdjson::dom::parser parser;
@@ -152,9 +152,9 @@ void sajson_traverse(stat_t &stats, const sajson::value &node) {
}
}
__attribute__((noinline)) stat_t
never_inline stat_t
sasjon_compute_stats(const simdjson::padded_string &p) {
stat_t answer;
stat_t answer{};
char *buffer = (char *)malloc(p.size());
if (buffer == nullptr) {
return answer;
@@ -214,9 +214,9 @@ void rapid_traverse(stat_t &stats, const rapidjson::Value &v) {
}
}
__attribute__((noinline)) stat_t
never_inline stat_t
rapid_compute_stats(const simdjson::padded_string &p) {
stat_t answer;
stat_t answer{};
char *buffer = (char *)malloc(p.size() + 1);
if (buffer == nullptr) {
return answer;
@@ -241,9 +241,9 @@ rapid_compute_stats(const simdjson::padded_string &p) {
return answer;
}
__attribute__((noinline)) stat_t
never_inline stat_t
rapid_accurate_compute_stats(const simdjson::padded_string &p) {
stat_t answer;
stat_t answer{};
char *buffer = (char *)malloc(p.size() + 1);
if (buffer == nullptr) {
return answer;
+13 -6
View File
@@ -65,17 +65,21 @@ bool fastjson_parse(const char *input) {
// end of fastjson stuff
#endif
never_inline size_t sum_line_lengths(char * data, size_t length) {
std::stringstream is;
is.rdbuf()->pubsetbuf(data, length);
never_inline size_t sum_line_lengths(std::stringstream & is) {
std::string line;
size_t sumofalllinelengths{0};
while(getline(is, line)) {
while(std::getline(is, line)) {
sumofalllinelengths += line.size();
}
return sumofalllinelengths;
}
inline void reset_stream(std::stringstream & is) {
is.clear();
is.seekg(0,std::ios::beg);
}
bool bench(const char *filename, bool verbose, bool just_data, double repeat_multiplier) {
auto [p, err] = simdjson::padded_string::load(filename);
@@ -103,8 +107,11 @@ bool bench(const char *filename, bool verbose, bool just_data, double repeat_mul
"cycles_per_byte_err", "gb_per_s", "gb_per_s_err");
}
if (!just_data) {
size_t lc = sum_line_lengths(p.data(), p.size());
BEST_TIME("getline ",sum_line_lengths(p.data(), p.size()) , lc, ,
const std::string inputcopy(p.data(), p.data()+p.size());
std::stringstream is;
is.str(inputcopy);
const size_t lc = sum_line_lengths(is);
BEST_TIME("getline ",sum_line_lengths(is) , lc, reset_stream(is),
repeat, volume, !just_data);
}
+2 -2
View File
@@ -5,12 +5,12 @@ endif()
if(MSVC)
option(SIMDJSON_BUILD_STATIC "Build a static library" ON) # turning it on disables the production of a dynamic library
option(SIMDJSON_COMPETITION "Compile competitive benchmarks" OFF)
else()
option(SIMDJSON_BUILD_STATIC "Build a static library" OFF) # turning it on disables the production of a dynamic library
option(SIMDJSON_COMPETITION "Compile competitive benchmarks" ON)
option(SIMDJSON_USE_LIBCPP "Use the libc++ library" OFF)
endif()
option(SIMDJSON_COMPETITION "Compile competitive benchmarks" ON)
option(SIMDJSON_GOOGLE_BENCHMARKS "compile the Google Benchmark benchmarks" ON)
set(CMAKE_MODULE_PATH "${CMAKE_CURRENT_SOURCE_DIR}/tools/cmake")