mirror of
https://github.com/simdjson/simdjson
synced 2026-06-08 17:27:07 +00:00
Compare commits
6 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| b2d39e9cda | |||
| 7662d7e57f | |||
| fad60f696a | |||
| 0507b42cb1 | |||
| b727c02950 | |||
| 47f28b88d6 |
@@ -13,20 +13,18 @@ jobs:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
include:
|
||||
- {gen: Visual Studio 17 2022, arch: Win32, shared: ON, build_type: Release, memory_map: OFF}
|
||||
- {gen: Visual Studio 17 2022, arch: Win32, shared: OFF, build_type: Release, memory_map: OFF}
|
||||
- {gen: Visual Studio 17 2022, arch: x64, shared: ON, build_type: Release, memory_map: OFF}
|
||||
- {gen: Visual Studio 17 2022, arch: x64, shared: OFF, build_type: Debug, memory_map: OFF}
|
||||
- {gen: Visual Studio 17 2022, arch: x64, shared: OFF, build_type: Release, memory_map: OFF}
|
||||
- {gen: Visual Studio 17 2022, arch: x64, shared: OFF, build_type: RelWithDebInfo, memory_map: OFF}
|
||||
# Exercise the opt-in Windows memory-file mapping path at least once in CI.
|
||||
- {gen: Visual Studio 17 2022, arch: x64, shared: OFF, build_type: Release, memory_map: ON}
|
||||
- {gen: Visual Studio 17 2022, arch: Win32, shared: ON, build_type: Release}
|
||||
- {gen: Visual Studio 17 2022, arch: Win32, shared: OFF, build_type: Release}
|
||||
- {gen: Visual Studio 17 2022, arch: x64, shared: ON, build_type: Release}
|
||||
- {gen: Visual Studio 17 2022, arch: x64, shared: OFF, build_type: Debug}
|
||||
- {gen: Visual Studio 17 2022, arch: x64, shared: OFF, build_type: Release}
|
||||
- {gen: Visual Studio 17 2022, arch: x64, shared: OFF, build_type: RelWithDebInfo}
|
||||
steps:
|
||||
- name: checkout
|
||||
uses: actions/checkout@v4
|
||||
- name: Configure
|
||||
run: |
|
||||
cmake -G "${{matrix.gen}}" -A ${{matrix.arch}} -DSIMDJSON_DEVELOPER_MODE=ON -DSIMDJSON_COMPETITION=OFF -DBUILD_SHARED_LIBS=${{matrix.shared}} -DSIMDJSON_ENABLE_MEMORY_FILE_MAPPING_ON_WINDOWS=${{matrix.memory_map}} -B build
|
||||
cmake -G "${{matrix.gen}}" -A ${{matrix.arch}} -DSIMDJSON_DEVELOPER_MODE=ON -DSIMDJSON_COMPETITION=OFF -DBUILD_SHARED_LIBS=${{matrix.shared}} -B build
|
||||
- name: Build Debug
|
||||
run: cmake --build build --config ${{matrix.build_type}} --verbose
|
||||
- name: Run tests
|
||||
|
||||
@@ -75,41 +75,6 @@ if(SIMDJSON_DEVELOPMENT_CHECKS)
|
||||
)
|
||||
endif()
|
||||
|
||||
# padded_memory_map is always available on POSIX. On Windows it is disabled
|
||||
# by default because it depends on the `CreateFileMapping2` / `MapViewOfFile3`
|
||||
# APIs, which require Windows 10 version 1803 or later and are exported via
|
||||
# onecore.lib rather than the default kernel32.lib. Turn this option ON to
|
||||
# opt into the feature on Windows; simdjson will then set the appropriate
|
||||
# Windows version macros and link onecore, so everything that links
|
||||
# simdjson picks up both the compile-time declarations and the import
|
||||
# library automatically. The option is a no-op on POSIX (where the feature
|
||||
# is unconditionally enabled).
|
||||
option(SIMDJSON_ENABLE_MEMORY_FILE_MAPPING_ON_WINDOWS
|
||||
"Enable simdjson::padded_memory_map on Windows (requires Windows 10 \
|
||||
version 1803 or later). Always enabled on POSIX." OFF)
|
||||
if(SIMDJSON_ENABLE_MEMORY_FILE_MAPPING_ON_WINDOWS)
|
||||
simdjson_add_props(
|
||||
target_compile_definitions PUBLIC
|
||||
SIMDJSON_ENABLE_MEMORY_FILE_MAPPING_ON_WINDOWS=1
|
||||
)
|
||||
if(WIN32)
|
||||
# Raise the Windows version floor so that <windows.h> declares the
|
||||
# modern memory-mapping APIs, and link the import library that
|
||||
# actually exports them. _WIN32_WINNT / WINVER / NTDDI_VERSION together
|
||||
# tell <sdkddkver.h> which APIs to light up.
|
||||
simdjson_add_props(
|
||||
target_compile_definitions PUBLIC
|
||||
_WIN32_WINNT=0x0A00
|
||||
WINVER=0x0A00
|
||||
NTDDI_VERSION=0x0A000006 # NTDDI_WIN10_RS5, Windows 10 version 1809
|
||||
)
|
||||
simdjson_add_props(
|
||||
target_link_libraries PUBLIC
|
||||
onecore
|
||||
)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
if(is_top_project)
|
||||
option(SIMDJSON_INSTALL "Enable target install" ON)
|
||||
option(SIMDJSON_DEVELOPER_MODE "Enable targets for developing simdjson" OFF)
|
||||
|
||||
@@ -1,10 +1,9 @@
|
||||
add_subdirectory(dom)
|
||||
|
||||
|
||||
include_directories( . )
|
||||
include_directories( . linux )
|
||||
link_libraries(simdjson-windows-headers test-data)
|
||||
link_libraries(simdjson)
|
||||
link_libraries(counters)
|
||||
if(SIMDJSON_STATIC_REFLECTION)
|
||||
add_compile_definitions(SIMDJSON_STATIC_REFLECTION=1)
|
||||
endif(SIMDJSON_STATIC_REFLECTION)
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -40,7 +40,6 @@ SIMDJSON_POP_DISABLE_WARNINGS
|
||||
#include "json2msgpack/boostjson.h"
|
||||
|
||||
#include "partial_tweets/simdjson_ondemand.h"
|
||||
#include "partial_tweets/simdjson_ondemand_key_selector.h"
|
||||
#include "partial_tweets/simdjson_dom.h"
|
||||
#include "partial_tweets/yyjson.h"
|
||||
#if SIMDJSON_COMPETITION_ONDEMAND_SAJSON
|
||||
@@ -125,6 +124,7 @@ SIMDJSON_POP_DISABLE_WARNINGS
|
||||
#include "kostya/boostjson.h"
|
||||
|
||||
#include "large_random/simdjson_ondemand.h"
|
||||
#include "large_random/simdjson_ondemand_ranges.h"
|
||||
#if SIMDJSON_COMPETITION_ONDEMAND_UNORDERED
|
||||
#include "large_random/simdjson_ondemand_unordered.h"
|
||||
#endif // SIMDJSON_COMPETITION_ONDEMAND_UNORDERED
|
||||
|
||||
@@ -1,5 +1,4 @@
|
||||
#include <counters/event_counter.h>
|
||||
using namespace counters;
|
||||
#include "event_counter.h"
|
||||
|
||||
#include <cassert>
|
||||
#include <cctype>
|
||||
@@ -26,6 +25,7 @@ using namespace counters;
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
#include "linux-perf-events.h"
|
||||
#ifdef __linux__
|
||||
#include <libgen.h>
|
||||
#endif
|
||||
@@ -204,8 +204,12 @@ struct feature_benchmarker {
|
||||
}
|
||||
// Rate of 1-7-structural misses per 8-structural flip
|
||||
double struct1_7_miss_rate(BenchmarkStage stage) const {
|
||||
#if SIMDJSON_SIMPLE_PERFORMANCE_COUNTERS
|
||||
return 1;
|
||||
#else
|
||||
if (!has_events()) { return 1; }
|
||||
return struct7_miss[stage].best.branch_misses() - struct7[stage].best.branch_misses() / double(struct7_miss.stats->blocks_with_1_structural_flipped);
|
||||
#endif
|
||||
}
|
||||
// Extra cost of an 8-15 structural block over a 1-7 structural block
|
||||
double struct8_15_cost(BenchmarkStage stage) const {
|
||||
@@ -217,8 +221,12 @@ struct feature_benchmarker {
|
||||
}
|
||||
// Rate of 8-15-structural misses per 8-structural flip
|
||||
double struct8_15_miss_rate(BenchmarkStage stage) const {
|
||||
#if SIMDJSON_SIMPLE_PERFORMANCE_COUNTERS
|
||||
return 1;
|
||||
#else
|
||||
if (!has_events()) { return 1; }
|
||||
return double(struct15_miss[stage].best.branch_misses() - struct15[stage].best.branch_misses()) / double(struct15_miss.stats->blocks_with_8_structurals_flipped);
|
||||
#endif
|
||||
}
|
||||
|
||||
// Extra cost of a 16+-structural block over an 8-15 structural block (actual varies based on # of structurals!)
|
||||
@@ -231,8 +239,12 @@ struct feature_benchmarker {
|
||||
}
|
||||
// Rate of 16-structural misses per 16-structural flip
|
||||
double struct16_miss_rate(BenchmarkStage stage) const {
|
||||
#if SIMDJSON_SIMPLE_PERFORMANCE_COUNTERS
|
||||
return 1;
|
||||
#else
|
||||
if (!has_events()) { return 1; }
|
||||
return double(struct23_miss[stage].best.branch_misses() - struct23[stage].best.branch_misses()) / double(struct23_miss.stats->blocks_with_16_structurals_flipped);
|
||||
#endif
|
||||
}
|
||||
|
||||
|
||||
@@ -246,8 +258,12 @@ struct feature_benchmarker {
|
||||
}
|
||||
// Rate of UTF-8 misses per UTF-8 flip
|
||||
double utf8_miss_rate(BenchmarkStage stage) const {
|
||||
#if SIMDJSON_SIMPLE_PERFORMANCE_COUNTERS
|
||||
return 1;
|
||||
#else
|
||||
if (!has_events()) { return 1; }
|
||||
return double(utf8_miss[stage].best.branch_misses() - utf8[stage].best.branch_misses()) / double(utf8_miss.stats->blocks_with_utf8_flipped);
|
||||
#endif
|
||||
}
|
||||
// Extra cost of having escapes in a block
|
||||
double escape_cost(BenchmarkStage stage) const {
|
||||
@@ -259,8 +275,12 @@ struct feature_benchmarker {
|
||||
}
|
||||
// Rate of escape misses per escape flip
|
||||
double escape_miss_rate(BenchmarkStage stage) const {
|
||||
#if SIMDJSON_SIMPLE_PERFORMANCE_COUNTERS
|
||||
return 1;
|
||||
#else
|
||||
if (!has_events()) { return 1; }
|
||||
return double(escape_miss[stage].best.branch_misses() - escape[stage].best.branch_misses()) / double(escape_miss.stats->blocks_with_escapes_flipped);
|
||||
#endif
|
||||
}
|
||||
|
||||
|
||||
@@ -358,6 +378,22 @@ struct feature_benchmarker {
|
||||
}
|
||||
};
|
||||
|
||||
#if SIMDJSON_SIMPLE_PERFORMANCE_COUNTERS
|
||||
void print_file_effectiveness(BenchmarkStage stage, const char* filename, const benchmarker& results, const feature_benchmarker& features) {
|
||||
double actual = results[stage].best.elapsed_ns() / double(results.stats->blocks);
|
||||
double calc = features.calc_expected(stage, results);
|
||||
double calc_misses = features.calc_expected_misses(stage, results);
|
||||
double calc_miss_cost = features.calc_expected_miss_cost(stage, results);
|
||||
printf(" | %-8s ", benchmark_stage_name(stage));
|
||||
printf("| %-15s ", filename);
|
||||
printf("| %8.3g ", features.calc_expected_feature_cost(stage, results));
|
||||
printf("| %8.3g ", calc_miss_cost);
|
||||
printf("| %8.3g ", calc);
|
||||
printf("| %8.3g ", actual);
|
||||
printf("| %+8.3g ", actual - calc);
|
||||
printf("| %13llu ", (long long unsigned)(calc_misses));
|
||||
}
|
||||
#else
|
||||
void print_file_effectiveness(BenchmarkStage stage, const char* filename, const benchmarker& results, const feature_benchmarker& features) {
|
||||
double actual = results[stage].best.elapsed_ns() / double(results.stats->blocks);
|
||||
double calc = features.calc_expected(stage, results);
|
||||
@@ -381,6 +417,7 @@ void print_file_effectiveness(BenchmarkStage stage, const char* filename, const
|
||||
}
|
||||
printf("|\n");
|
||||
}
|
||||
#endif
|
||||
|
||||
int main(int argc, char *argv[]) {
|
||||
// Read options
|
||||
|
||||
@@ -1,8 +1,7 @@
|
||||
#ifndef _BENCHMARK_H_
|
||||
#define _BENCHMARK_H_
|
||||
|
||||
#include <counters/event_counter.h>
|
||||
using namespace counters;
|
||||
#include "event_counter.h"
|
||||
|
||||
/*
|
||||
* Prints the best number of operations per cycle where
|
||||
|
||||
+11
-4
@@ -1,8 +1,7 @@
|
||||
#ifndef __BENCHMARKER_H
|
||||
#define __BENCHMARKER_H
|
||||
|
||||
#include <counters/event_counter.h>
|
||||
using namespace counters;
|
||||
#include "event_counter.h"
|
||||
#include "simdjson.h"
|
||||
|
||||
#include <cassert>
|
||||
@@ -29,9 +28,11 @@ using namespace counters;
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
#include "linux-perf-events.h"
|
||||
#ifdef __linux__
|
||||
#include <libgen.h>
|
||||
#endif
|
||||
#include "simdjson.h"
|
||||
|
||||
#include <functional>
|
||||
|
||||
@@ -422,12 +423,18 @@ struct benchmarker {
|
||||
stage.instructions() / static_cast<double>(stats->structurals),
|
||||
stage.instructions() / static_cast<double>(stage.cycles())
|
||||
);
|
||||
printf("%s%-13s: %7.0f branch misses (%6.2f%%)\n",
|
||||
#if !SIMDJSON_SIMPLE_PERFORMANCE_COUNTERS
|
||||
// NOTE: removed cycles/miss because it is a somewhat misleading stat
|
||||
printf("%s%-13s: %7.0f branch misses (%6.2f%%) - %.0f cache misses (%6.2f%%) - %.2f cache references\n",
|
||||
prefix,
|
||||
"Misses",
|
||||
stage.branch_misses(),
|
||||
percent(stage.branch_misses(), all_stages_without_allocation.branch_misses())
|
||||
percent(stage.branch_misses(), all_stages_without_allocation.branch_misses()),
|
||||
stage.cache_misses(),
|
||||
percent(stage.cache_misses(), all_stages_without_allocation.cache_misses()),
|
||||
stage.cache_references()
|
||||
);
|
||||
#endif
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -1,5 +1,4 @@
|
||||
#include <counters/event_counter.h>
|
||||
using namespace counters;
|
||||
#include "event_counter.h"
|
||||
#include <random>
|
||||
#include <vector>
|
||||
|
||||
|
||||
@@ -1,12 +1,11 @@
|
||||
include_directories( .. )
|
||||
include_directories( .. ../linux )
|
||||
link_libraries(simdjson-windows-headers test-data)
|
||||
link_libraries(simdjson)
|
||||
link_libraries(counters)
|
||||
|
||||
add_executable(perfdiff perfdiff.cpp)
|
||||
add_executable(parse parse.cpp)
|
||||
add_executable(parse_stream parse_stream.cpp)
|
||||
|
||||
add_executable(statisticalmodel statisticalmodel.cpp)
|
||||
|
||||
add_executable(parse_noutf8validation parse.cpp)
|
||||
target_compile_definitions(parse_noutf8validation PRIVATE SIMDJSON_SKIPUTF8VALIDATION)
|
||||
|
||||
@@ -1,5 +1,4 @@
|
||||
#include <counters/event_counter.h>
|
||||
using namespace counters;
|
||||
#include "event_counter.h"
|
||||
|
||||
#include <cassert>
|
||||
#include <cctype>
|
||||
@@ -25,6 +24,7 @@ using namespace counters;
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
#include "linux-perf-events.h"
|
||||
#ifdef __linux__
|
||||
#include <libgen.h>
|
||||
#endif
|
||||
|
||||
@@ -0,0 +1,207 @@
|
||||
#include <iostream>
|
||||
#include <unistd.h>
|
||||
#include "simdjson.h"
|
||||
#ifdef __linux__
|
||||
#include "linux-perf-events.h"
|
||||
#endif
|
||||
|
||||
size_t count_nonasciibytes(const uint8_t *input, size_t length) {
|
||||
size_t count = 0;
|
||||
for (size_t i = 0; i < length; i++) {
|
||||
count += input[i] >> 7;
|
||||
}
|
||||
return count;
|
||||
}
|
||||
|
||||
size_t count_backslash(const uint8_t *input, size_t length) {
|
||||
size_t count = 0;
|
||||
for (size_t i = 0; i < length; i++) {
|
||||
count += (input[i] == '\\') ? 1 : 0;
|
||||
}
|
||||
return count;
|
||||
}
|
||||
|
||||
struct stat_s {
|
||||
size_t integer_count;
|
||||
size_t float_count;
|
||||
size_t string_count;
|
||||
size_t backslash_count;
|
||||
size_t non_ascii_byte_count;
|
||||
size_t object_count;
|
||||
size_t array_count;
|
||||
size_t null_count;
|
||||
size_t true_count;
|
||||
size_t false_count;
|
||||
size_t byte_count;
|
||||
size_t structural_indexes_count;
|
||||
bool valid;
|
||||
};
|
||||
|
||||
using stat_t = struct stat_s;
|
||||
|
||||
|
||||
|
||||
simdjson_inline void simdjson_process_atom(stat_t &s,
|
||||
simdjson::dom::element element) {
|
||||
if (element.is<int64_t>()) {
|
||||
s.integer_count++;
|
||||
} else if(element.is<std::string_view>()) {
|
||||
s.string_count++;
|
||||
} else if(element.is<double>()) {
|
||||
s.float_count++;
|
||||
} else if (element.is<bool>()) {
|
||||
bool v;
|
||||
simdjson::error_code error;
|
||||
if ((error = element.get(v))) { std::cerr << error << std::endl; abort(); }
|
||||
if (v) {
|
||||
s.true_count++;
|
||||
} else {
|
||||
s.false_count++;
|
||||
}
|
||||
} else if (element.is_null()) {
|
||||
s.null_count++;
|
||||
}
|
||||
}
|
||||
|
||||
void simdjson_recurse(stat_t &s, simdjson::dom::element element) {
|
||||
simdjson::error_code error;
|
||||
if (element.is<simdjson::dom::array>()) {
|
||||
s.array_count++;
|
||||
simdjson::dom::array array;
|
||||
if ((error = element.get(array))) { std::cerr << error << std::endl; abort(); }
|
||||
for (auto child : array) {
|
||||
if (child.is<simdjson::dom::array>() || child.is<simdjson::dom::object>()) {
|
||||
simdjson_recurse(s, child);
|
||||
} else {
|
||||
simdjson_process_atom(s, child);
|
||||
}
|
||||
}
|
||||
} else if (element.is<simdjson::dom::object>()) {
|
||||
s.object_count++;
|
||||
simdjson::dom::object object;
|
||||
if ((error = element.get(object))) { std::cerr << error << std::endl; abort(); }
|
||||
for (auto field : object) {
|
||||
s.string_count++; // for key
|
||||
if (field.value.is<simdjson::dom::array>() || field.value.is<simdjson::dom::object>()) {
|
||||
simdjson_recurse(s, field.value);
|
||||
} else {
|
||||
simdjson_process_atom(s, field.value);
|
||||
}
|
||||
}
|
||||
} else {
|
||||
simdjson_process_atom(s, element);
|
||||
}
|
||||
}
|
||||
|
||||
stat_t simdjson_compute_stats(const simdjson::padded_string &p) {
|
||||
stat_t answer{};
|
||||
simdjson::dom::parser parser;
|
||||
simdjson::dom::element doc;
|
||||
auto error = parser.parse(p).get(doc);
|
||||
if (error) {
|
||||
answer.valid = false;
|
||||
return answer;
|
||||
}
|
||||
answer.valid = true;
|
||||
answer.backslash_count =
|
||||
count_backslash(reinterpret_cast<const uint8_t *>(p.data()), p.size());
|
||||
answer.non_ascii_byte_count = count_nonasciibytes(
|
||||
reinterpret_cast<const uint8_t *>(p.data()), p.size());
|
||||
answer.byte_count = p.size();
|
||||
answer.structural_indexes_count = parser.implementation->n_structural_indexes;
|
||||
simdjson_recurse(answer, doc);
|
||||
return answer;
|
||||
}
|
||||
|
||||
int main(int argc, char *argv[]) {
|
||||
#ifndef _MSC_VER
|
||||
int c;
|
||||
while ((c = getopt(argc, argv, "")) != -1) {
|
||||
switch (c) {
|
||||
|
||||
default:
|
||||
abort();
|
||||
}
|
||||
}
|
||||
#else
|
||||
int optind = 1;
|
||||
#endif
|
||||
if (optind >= argc) {
|
||||
std::cerr << "Reads json, prints stats. " << std::endl;
|
||||
std::cerr << "Usage: " << argv[0] << " <jsonfile>" << std::endl;
|
||||
|
||||
exit(1);
|
||||
}
|
||||
const char *filename = argv[optind];
|
||||
if (optind + 1 < argc) {
|
||||
std::cerr << "warning: ignoring everything after " << argv[optind + 1]
|
||||
<< std::endl;
|
||||
}
|
||||
simdjson::padded_string p;
|
||||
auto error = simdjson::padded_string::load(filename).get(p);
|
||||
if (error) {
|
||||
std::cerr << "Could not load the file " << filename << std::endl;
|
||||
return EXIT_FAILURE;
|
||||
}
|
||||
stat_t s = simdjson_compute_stats(p);
|
||||
if (!s.valid) {
|
||||
std::cerr << "not a valid JSON" << std::endl;
|
||||
return EXIT_FAILURE;
|
||||
}
|
||||
|
||||
printf("# integer_count float_count string_count backslash_count "
|
||||
"non_ascii_byte_count object_count array_count null_count true_count "
|
||||
"false_count byte_count structural_indexes_count ");
|
||||
#ifdef __linux__
|
||||
printf(" stage1_cycle_count stage1_instruction_count stage2_cycle_count "
|
||||
" stage2_instruction_count stage3_cycle_count "
|
||||
"stage3_instruction_count ");
|
||||
#else
|
||||
printf("(you are not under linux, so perf counters are disaabled)");
|
||||
#endif
|
||||
printf("\n");
|
||||
printf("%zu %zu %zu %zu %zu %zu %zu %zu %zu %zu %zu %zu ", s.integer_count,
|
||||
s.float_count, s.string_count, s.backslash_count,
|
||||
s.non_ascii_byte_count, s.object_count, s.array_count, s.null_count,
|
||||
s.true_count, s.false_count, s.byte_count, s.structural_indexes_count);
|
||||
#ifdef __linux__
|
||||
simdjson::dom::parser parser;
|
||||
simdjson::error_code alloc_error = parser.allocate(p.size());
|
||||
if (alloc_error) {
|
||||
std::cerr << alloc_error << std::endl;
|
||||
return EXIT_FAILURE;
|
||||
}
|
||||
const uint32_t iterations = p.size() < 1 * 1000 * 1000 ? 1000 : 50;
|
||||
std::vector<int> evts;
|
||||
evts.push_back(PERF_COUNT_HW_CPU_CYCLES);
|
||||
evts.push_back(PERF_COUNT_HW_INSTRUCTIONS);
|
||||
LinuxEvents<PERF_TYPE_HARDWARE> unified(evts);
|
||||
unsigned long cy1 = 0, cy2 = 0;
|
||||
unsigned long cl1 = 0, cl2 = 0;
|
||||
std::vector<unsigned long long> results;
|
||||
results.resize(evts.size());
|
||||
for (uint32_t i = 0; i < iterations; i++) {
|
||||
unified.start();
|
||||
// The default template is simdjson::architecture::NATIVE.
|
||||
bool isok = (parser.implementation->stage1((const uint8_t *)p.data(), p.size(), simdjson::stage1_mode::regular) == simdjson::SUCCESS);
|
||||
unified.end(results);
|
||||
|
||||
cy1 += results[0];
|
||||
cl1 += results[1];
|
||||
|
||||
unified.start();
|
||||
isok = isok && (parser.implementation->stage2(parser.doc) == simdjson::SUCCESS);
|
||||
unified.end(results);
|
||||
|
||||
cy2 += results[0];
|
||||
cl2 += results[1];
|
||||
if (!isok) {
|
||||
std::cerr << "failure?" << std::endl;
|
||||
}
|
||||
}
|
||||
printf("%f %f %f %f ", static_cast<double>(cy1) / static_cast<double>(iterations), static_cast<double>(cl1) / static_cast<double>(iterations),
|
||||
static_cast<double>(cy2) / static_cast<double>(iterations), static_cast<double>(cl2) / static_cast<double>(iterations));
|
||||
#endif // __linux__
|
||||
printf("\n");
|
||||
return EXIT_SUCCESS;
|
||||
}
|
||||
@@ -0,0 +1,201 @@
|
||||
#ifndef __EVENT_COUNTER_H
|
||||
#define __EVENT_COUNTER_H
|
||||
|
||||
#ifndef SIMDJSON_SIMPLE_PERFORMANCE_COUNTERS
|
||||
#ifdef __aarch64__
|
||||
// on ARM, we use just cycles and instructions
|
||||
#define SIMDJSON_SIMPLE_PERFORMANCE_COUNTERS 1
|
||||
#else
|
||||
// elsewhere, we try to use four counters.
|
||||
#define SIMDJSON_SIMPLE_PERFORMANCE_COUNTERS 0
|
||||
#endif
|
||||
#endif
|
||||
#include <cassert>
|
||||
#include <cctype>
|
||||
#ifndef _MSC_VER
|
||||
#include <dirent.h>
|
||||
#endif
|
||||
#include <unistd.h>
|
||||
#include <cinttypes>
|
||||
|
||||
#include <cstdio>
|
||||
#include <cstdlib>
|
||||
#include <cstring>
|
||||
|
||||
#include <algorithm>
|
||||
#include <chrono>
|
||||
#include <cstring>
|
||||
#include <fstream>
|
||||
#include <iomanip>
|
||||
#include <iostream>
|
||||
#include <map>
|
||||
#include <set>
|
||||
#include <sstream>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
#ifdef __linux__
|
||||
#include "linux-perf-events.h"
|
||||
#include <libgen.h>
|
||||
#endif
|
||||
|
||||
#if __APPLE__ && __aarch64__
|
||||
#include "apple/apple_arm_events.h"
|
||||
#endif
|
||||
|
||||
#include "simdjson.h"
|
||||
|
||||
using std::string;
|
||||
using std::vector;
|
||||
using std::chrono::steady_clock;
|
||||
using std::chrono::time_point;
|
||||
using std::chrono::duration;
|
||||
|
||||
struct event_count {
|
||||
duration<double> elapsed;
|
||||
vector<unsigned long long> event_counts;
|
||||
event_count() : elapsed(0), event_counts{0,0,0,0,0} {}
|
||||
event_count(const duration<double> _elapsed, const vector<unsigned long long> _event_counts) : elapsed(_elapsed), event_counts(_event_counts) {}
|
||||
event_count(const event_count& other): elapsed(other.elapsed), event_counts(other.event_counts) { }
|
||||
|
||||
// The types of counters (so we can read the getter more easily)
|
||||
#if SIMDJSON_SIMPLE_PERFORMANCE_COUNTERS
|
||||
enum event_counter_types {
|
||||
CPU_CYCLES,
|
||||
INSTRUCTIONS
|
||||
};
|
||||
#else
|
||||
enum event_counter_types {
|
||||
CPU_CYCLES,
|
||||
INSTRUCTIONS,
|
||||
BRANCH_MISSES,
|
||||
CACHE_REFERENCES,
|
||||
CACHE_MISSES
|
||||
};
|
||||
#endif
|
||||
double elapsed_sec() const { return duration<double>(elapsed).count(); }
|
||||
double elapsed_ns() const { return duration<double, std::nano>(elapsed).count(); }
|
||||
double cycles() const { return static_cast<double>(event_counts[CPU_CYCLES]); }
|
||||
double instructions() const { return static_cast<double>(event_counts[INSTRUCTIONS]); }
|
||||
#if !SIMDJSON_SIMPLE_PERFORMANCE_COUNTERS
|
||||
double branch_misses() const { return static_cast<double>(event_counts[BRANCH_MISSES]); }
|
||||
double cache_references() const { return static_cast<double>(event_counts[CACHE_REFERENCES]); }
|
||||
double cache_misses() const { return static_cast<double>(event_counts[CACHE_MISSES]); }
|
||||
#endif
|
||||
event_count& operator=(const event_count& other) {
|
||||
this->elapsed = other.elapsed;
|
||||
this->event_counts = other.event_counts;
|
||||
return *this;
|
||||
}
|
||||
event_count operator+(const event_count& other) const {
|
||||
return event_count(elapsed+other.elapsed, {
|
||||
event_counts[0]+other.event_counts[0],
|
||||
event_counts[1]+other.event_counts[1],
|
||||
event_counts[2]+other.event_counts[2],
|
||||
event_counts[3]+other.event_counts[3],
|
||||
event_counts[4]+other.event_counts[4],
|
||||
});
|
||||
}
|
||||
|
||||
void operator+=(const event_count& other) {
|
||||
*this = *this + other;
|
||||
}
|
||||
};
|
||||
|
||||
struct event_aggregate {
|
||||
int iterations = 0;
|
||||
event_count total{};
|
||||
event_count best{};
|
||||
event_count worst{};
|
||||
|
||||
event_aggregate() {}
|
||||
|
||||
void operator<<(const event_count& other) {
|
||||
if (iterations == 0 || other.elapsed < best.elapsed) {
|
||||
best = other;
|
||||
}
|
||||
if (iterations == 0 || other.elapsed > worst.elapsed) {
|
||||
worst = other;
|
||||
}
|
||||
iterations++;
|
||||
total += other;
|
||||
}
|
||||
|
||||
double elapsed_sec() const { return total.elapsed_sec() / iterations; }
|
||||
double total_elapsed_ns() const { return total.elapsed_ns(); }
|
||||
double elapsed_ns() const { return total.elapsed_ns() / iterations; }
|
||||
double cycles() const { return total.cycles() / iterations; }
|
||||
double instructions() const { return total.instructions() / iterations; }
|
||||
#if !SIMDJSON_SIMPLE_PERFORMANCE_COUNTERS
|
||||
double branch_misses() const { return total.branch_misses() / iterations; }
|
||||
double cache_references() const { return total.cache_references() / iterations; }
|
||||
double cache_misses() const { return total.cache_misses() / iterations; }
|
||||
#endif
|
||||
};
|
||||
|
||||
struct event_collector {
|
||||
event_count count{};
|
||||
time_point<steady_clock> start_clock{};
|
||||
|
||||
#if defined(__linux__)
|
||||
LinuxEvents<PERF_TYPE_HARDWARE> linux_events;
|
||||
event_collector() : linux_events(vector<int>{
|
||||
#if SIMDJSON_SIMPLE_PERFORMANCE_COUNTERS
|
||||
PERF_COUNT_HW_CPU_CYCLES,
|
||||
PERF_COUNT_HW_INSTRUCTIONS,
|
||||
#else
|
||||
PERF_COUNT_HW_CPU_CYCLES,
|
||||
PERF_COUNT_HW_INSTRUCTIONS,
|
||||
PERF_COUNT_HW_BRANCH_MISSES,
|
||||
PERF_COUNT_HW_CACHE_REFERENCES,
|
||||
PERF_COUNT_HW_CACHE_MISSES
|
||||
#endif
|
||||
}) {}
|
||||
bool has_events() {
|
||||
return linux_events.is_working();
|
||||
}
|
||||
#elif __APPLE__ && __aarch64__
|
||||
AppleEvents apple_events;
|
||||
performance_counters diff;
|
||||
event_collector() : diff(0) {
|
||||
apple_events.setup_performance_counters();
|
||||
}
|
||||
bool has_events() {
|
||||
return apple_events.setup_performance_counters();
|
||||
}
|
||||
#else
|
||||
event_collector() {}
|
||||
bool has_events() {
|
||||
return false;
|
||||
}
|
||||
#endif
|
||||
|
||||
simdjson_inline void start() {
|
||||
#if defined(__linux)
|
||||
linux_events.start();
|
||||
#elif __APPLE__ && __aarch64__
|
||||
if(has_events()) { diff = apple_events.get_counters(); }
|
||||
#endif
|
||||
start_clock = steady_clock::now();
|
||||
}
|
||||
simdjson_inline event_count& end() {
|
||||
time_point<steady_clock> end_clock = steady_clock::now();
|
||||
#if defined(__linux)
|
||||
linux_events.end(count.event_counts);
|
||||
#elif __APPLE__ && __aarch64__
|
||||
if(has_events()) {
|
||||
performance_counters end = apple_events.get_counters();
|
||||
diff = end - diff;
|
||||
}
|
||||
count.event_counts[0] = diff.cycles;
|
||||
count.event_counts[1] = diff.instructions;
|
||||
count.event_counts[2] = diff.missed_branches;
|
||||
count.event_counts[3] = 0;
|
||||
count.event_counts[4] = 0;
|
||||
#endif
|
||||
count.elapsed = end_clock - start_clock;
|
||||
return count;
|
||||
}
|
||||
};
|
||||
|
||||
#endif
|
||||
@@ -1,8 +1,7 @@
|
||||
#ifndef BENCHMARK_HELPERS_H
|
||||
#define BENCHMARK_HELPERS_H
|
||||
|
||||
#include <counters/event_counter.h>
|
||||
using namespace counters;
|
||||
#include "event_counter.h"
|
||||
#include <atomic>
|
||||
|
||||
event_collector collector;
|
||||
|
||||
@@ -1,8 +1,7 @@
|
||||
#pragma once
|
||||
|
||||
#include "simdjson.h"
|
||||
#include <counters/event_counter.h>
|
||||
using namespace counters;
|
||||
#include "event_counter.h"
|
||||
#include <iostream>
|
||||
|
||||
namespace json_benchmark {
|
||||
@@ -59,7 +58,11 @@ template<typename B, typename R> static void run_json_benchmark(benchmark::State
|
||||
if (collector.has_events()) {
|
||||
state.counters["instructions"] = events.instructions();
|
||||
state.counters["cycles"] = events.cycles();
|
||||
#if !SIMDJSON_SIMPLE_PERFORMANCE_COUNTERS
|
||||
state.counters["branch_miss"] = events.branch_misses();
|
||||
state.counters["cache_miss"] = events.cache_misses();
|
||||
state.counters["cache_ref"] = events.cache_references();
|
||||
#endif
|
||||
state.counters["instructions_per_byte"] = events.instructions() / double(bench.bytes_per_iteration());
|
||||
state.counters["instructions_per_cycle"] = events.instructions() / events.cycles();
|
||||
state.counters["cycles_per_byte"] = events.cycles() / double(bench.bytes_per_iteration());
|
||||
@@ -67,7 +70,11 @@ template<typename B, typename R> static void run_json_benchmark(benchmark::State
|
||||
|
||||
state.counters["best_instructions"] = events.best.instructions();
|
||||
state.counters["best_cycles"] = events.best.cycles();
|
||||
#if !SIMDJSON_SIMPLE_PERFORMANCE_COUNTERS
|
||||
state.counters["best_branch_miss"] = events.best.branch_misses();
|
||||
state.counters["best_cache_miss"] = events.best.cache_misses();
|
||||
state.counters["best_cache_ref"] = events.best.cache_references();
|
||||
#endif
|
||||
|
||||
state.counters["best_instructions_per_byte"] = events.best.instructions() / double(bench.bytes_per_iteration());
|
||||
state.counters["best_instructions_per_cycle"] = events.best.instructions() / events.best.cycles();
|
||||
@@ -88,7 +95,11 @@ template<typename B, typename R> static void run_json_benchmark(benchmark::State
|
||||
if (collector.has_events()) {
|
||||
label << " instructions=" << setw(12) << uint64_t(events.best.instructions()) << setw(0);
|
||||
label << " cycles=" << setw(12) << uint64_t(events.best.cycles()) << setw(0);
|
||||
#if !SIMDJSON_SIMPLE_PERFORMANCE_COUNTERS
|
||||
label << " branch_miss=" << setw(8) << uint64_t(events.best.branch_misses()) << setw(0);
|
||||
label << " cache_miss=" << setw(8) << uint64_t(events.best.cache_misses()) << setw(0);
|
||||
label << " cache_ref=" << setw(10) << uint64_t(events.best.cache_references()) << setw(0);
|
||||
#endif
|
||||
}
|
||||
|
||||
label << " items=" << setw(10) << bench.items_per_iteration() << setw(0);
|
||||
|
||||
@@ -1,7 +1,6 @@
|
||||
#pragma once
|
||||
|
||||
#include "json_benchmark/string_runner.h"
|
||||
#include <fstream>
|
||||
#include <map>
|
||||
#include <string>
|
||||
|
||||
|
||||
@@ -0,0 +1,32 @@
|
||||
#pragma once
|
||||
|
||||
#if SIMDJSON_EXCEPTIONS && SIMDJSON_SUPPORTS_RANGES
|
||||
|
||||
#include "large_random.h"
|
||||
|
||||
namespace large_random {
|
||||
|
||||
using namespace simdjson;
|
||||
|
||||
// Identical to simdjson_ondemand but uses get_range() for iteration.
|
||||
// Demonstrates that the ranges wrapper has zero per-element overhead.
|
||||
struct simdjson_ondemand_ranges {
|
||||
static constexpr diff_flags DiffFlags = diff_flags::NONE;
|
||||
|
||||
ondemand::parser parser{};
|
||||
|
||||
bool run(simdjson::padded_string &json, std::vector<point> &result) {
|
||||
auto doc = parser.iterate(json);
|
||||
for (auto coord_result : ondemand::get_range(doc.get_array())) {
|
||||
ondemand::object coord = coord_result;
|
||||
result.emplace_back(json_benchmark::point{coord.find_field("x"), coord.find_field("y"), coord.find_field("z")});
|
||||
}
|
||||
return true;
|
||||
}
|
||||
};
|
||||
|
||||
BENCHMARK_TEMPLATE(large_random, simdjson_ondemand_ranges)->UseManualTime();
|
||||
|
||||
} // namespace large_random
|
||||
|
||||
#endif // SIMDJSON_EXCEPTIONS && SIMDJSON_SUPPORTS_RANGES
|
||||
@@ -0,0 +1,105 @@
|
||||
#pragma once
|
||||
#ifdef __linux__
|
||||
|
||||
#include <asm/unistd.h> // for __NR_perf_event_open
|
||||
#include <linux/perf_event.h> // for perf event constants
|
||||
#include <sys/ioctl.h> // for ioctl
|
||||
#include <unistd.h> // for syscall
|
||||
|
||||
#include <cerrno> // for errno
|
||||
#include <cstring> // for memset
|
||||
#include <stdexcept>
|
||||
|
||||
#include <iostream>
|
||||
#include <vector>
|
||||
|
||||
template <int TYPE = PERF_TYPE_HARDWARE> class LinuxEvents {
|
||||
int fd;
|
||||
bool working;
|
||||
perf_event_attr attribs{};
|
||||
size_t num_events{};
|
||||
std::vector<uint64_t> temp_result_vec{};
|
||||
std::vector<uint64_t> ids{};
|
||||
|
||||
public:
|
||||
explicit LinuxEvents(std::vector<int> config_vec) : fd(0), working(true) {
|
||||
memset(&attribs, 0, sizeof(attribs));
|
||||
attribs.type = TYPE;
|
||||
attribs.size = sizeof(attribs);
|
||||
attribs.disabled = 1;
|
||||
attribs.exclude_kernel = 1;
|
||||
attribs.exclude_hv = 1;
|
||||
|
||||
attribs.sample_period = 0;
|
||||
attribs.read_format = PERF_FORMAT_GROUP | PERF_FORMAT_ID;
|
||||
const int pid = 0; // the current process
|
||||
const int cpu = -1; // all CPUs
|
||||
const unsigned long flags = 0;
|
||||
|
||||
int group = -1; // no group
|
||||
num_events = config_vec.size();
|
||||
ids.resize(config_vec.size());
|
||||
uint32_t i = 0;
|
||||
for (auto config : config_vec) {
|
||||
attribs.config = config;
|
||||
int _fd = static_cast<int>(syscall(__NR_perf_event_open, &attribs, pid, cpu, group, flags));
|
||||
if (_fd == -1) {
|
||||
report_error("perf_event_open");
|
||||
}
|
||||
ioctl(_fd, PERF_EVENT_IOC_ID, &ids[i++]);
|
||||
if (group == -1) {
|
||||
group = _fd;
|
||||
fd = _fd;
|
||||
}
|
||||
}
|
||||
|
||||
temp_result_vec.resize(num_events * 2 + 1);
|
||||
}
|
||||
|
||||
~LinuxEvents() { if (fd != -1) { close(fd); } }
|
||||
|
||||
inline void start() {
|
||||
if (fd != -1) {
|
||||
if (ioctl(fd, PERF_EVENT_IOC_RESET, PERF_IOC_FLAG_GROUP) == -1) {
|
||||
report_error("ioctl(PERF_EVENT_IOC_RESET)");
|
||||
}
|
||||
|
||||
if (ioctl(fd, PERF_EVENT_IOC_ENABLE, PERF_IOC_FLAG_GROUP) == -1) {
|
||||
report_error("ioctl(PERF_EVENT_IOC_ENABLE)");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
inline void end(std::vector<unsigned long long> &results) {
|
||||
if (fd != -1) {
|
||||
if (ioctl(fd, PERF_EVENT_IOC_DISABLE, PERF_IOC_FLAG_GROUP) == -1) {
|
||||
report_error("ioctl(PERF_EVENT_IOC_DISABLE)");
|
||||
}
|
||||
|
||||
if (read(fd, temp_result_vec.data(), temp_result_vec.size() * 8) == -1) {
|
||||
report_error("read");
|
||||
}
|
||||
}
|
||||
// our actual results are in slots 1,3,5, ... of this structure
|
||||
for (uint32_t i = 1; i < temp_result_vec.size(); i += 2) {
|
||||
results[i / 2] = temp_result_vec[i];
|
||||
}
|
||||
for (uint32_t i = 2; i < temp_result_vec.size(); i += 2) {
|
||||
if(ids[i/2-1] != temp_result_vec[i]) {
|
||||
report_error("event mismatch");
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
}
|
||||
|
||||
bool is_working() {
|
||||
return working;
|
||||
}
|
||||
|
||||
private:
|
||||
void report_error(const std::string &) {
|
||||
working = false;
|
||||
}
|
||||
};
|
||||
#endif
|
||||
@@ -1,7 +1,6 @@
|
||||
#ifndef BENCHMARK_HELPER_HPP
|
||||
#define BENCHMARK_HELPER_HPP
|
||||
#include <counters/event_counter.h>
|
||||
using namespace counters;
|
||||
#include "event_counter.h"
|
||||
#include <atomic>
|
||||
|
||||
inline event_collector &get_collector() {
|
||||
|
||||
Vendored
-8
@@ -20,14 +20,6 @@ if(SIMDJSON_GOOGLE_BENCHMARKS)
|
||||
)
|
||||
endif()
|
||||
|
||||
CPMAddPackage(
|
||||
NAME counters
|
||||
URL https://github.com/lemire/counters/archive/refs/tags/v3.1.0.zip
|
||||
OPTIONS
|
||||
"COUNTERS_BUILD_TESTS OFF"
|
||||
"COUNTERS_INSTALL OFF"
|
||||
)
|
||||
|
||||
CPMAddPackage(
|
||||
NAME simdjson-data
|
||||
URL https://github.com/simdjson/simdjson-data/archive/351949906abde446f0314bf79606fb5d884f5be7.zip
|
||||
|
||||
+63
-28
@@ -269,35 +269,10 @@ ondemand::document doc = parser.iterate(simdjson::pad(json));
|
||||
We recommend against creating many `std::string` or many `std::padded_string` instances in your application to store your JSON data.
|
||||
Consider reusing the same buffers and limiting memory allocations.
|
||||
|
||||
**Memory-file mapping.** You can use `simdjson::padded_memory_map` to create a
|
||||
`simdjson::padded_string_view` from a file on disk. On POSIX systems (Linux,
|
||||
macOS, BSD, ...) it uses `mmap` for true zero-copy access and is always
|
||||
available. On Windows it is an **opt-in** feature because it relies on the
|
||||
`CreateFileMapping2` / `MapViewOfFile3` APIs (Windows 10, version 1803 or
|
||||
later) which are exported from `onecore.lib` rather than the default
|
||||
`kernel32.lib`. To enable it, you must satisfy **all** of the following:
|
||||
|
||||
1. Building simdjson with `-DSIMDJSON_ENABLE_MEMORY_FILE_MAPPING_ON_WINDOWS=ON`, or
|
||||
defining `SIMDJSON_ENABLE_MEMORY_FILE_MAPPING_ON_WINDOWS=1` and raising
|
||||
`NTDDI_VERSION` to at least `NTDDI_WIN10_RS4` (Windows 10, version 1803)
|
||||
and linking `onecore.lib` manually if you are consuming simdjson as a
|
||||
pre-built library.
|
||||
2. `#include <windows.h>` before including simdjson, in every translation
|
||||
unit that uses `padded_memory_map`.
|
||||
|
||||
The Windows implementation then uses `CreateFileMapping2` / `MapViewOfFile3`
|
||||
for true zero-copy access whenever possible, with a transparent
|
||||
buffered-read fallback for files that end too close to a page boundary.
|
||||
|
||||
The availability of the class can be tested with the preprocessor macro
|
||||
`SIMDJSON_HAS_PADDED_MEMORY_MAP`.
|
||||
**Memory-file mapping (non-Windows).** You can use memory-file mapping to create a `simdjson::padded_string_view`
|
||||
from a file on disk:
|
||||
|
||||
```cpp
|
||||
#ifdef _WIN32
|
||||
#include <windows.h> // Must come BEFORE <simdjson.h> on Windows
|
||||
#endif
|
||||
#include "simdjson.h"
|
||||
// ...
|
||||
simdjson::padded_memory_map map(myfilename);
|
||||
if (!map.is_valid()) { /* handle error */ }
|
||||
simdjson::padded_string_view view = map.view();
|
||||
@@ -330,7 +305,7 @@ Some users may want to browse code along with the compiled assembly:
|
||||
| `simdjson::pad(std::string&)` | Adds padding if needed | Returns `padded_string_view` pointing to the (possibly resized) string | References original string | Recommended to silence sanitizers when using `std::string`. |
|
||||
| `padded_string(data, length)` or `padded_string(std::string)` | Automatic (copies into padded buffer) | Explicit copy into owned padded buffer | Owned by `padded_string` | Safe when you want full ownership and padding guaranteed. |
|
||||
| `padded_string_view` (manual) | User guarantees `SIMDJSON_PADDING` extra bytes after the viewed length | User provides pointer + length + capacity | Non-owning view | Low-level; requires careful buffer management. |
|
||||
| Memory-mapped file (`padded_memory_map`) | Automatic via mapping / padded read | Creates view with sufficient padding | Non-owning (tied to map lifetime) | Always available on POSIX (zero-copy `mmap`). On Windows, opt-in via `-DSIMDJSON_ENABLE_MEMORY_FILE_MAPPING_ON_WINDOWS=ON` (requires Windows 10 1803+ and links `onecore.lib`) and `#include <windows.h>` before simdjson; uses `CreateFileMapping2` + `MapViewOfFile3`. |
|
||||
| Memory-mapped file (`padded_memory_map`) | Automatic via mapping (non-Windows only) | Creates view with sufficient padding | Non-owning (tied to map lifetime) | Advanced; efficient for large files on Linux/macOS/etc. |
|
||||
|
||||
|
||||
Documents are iterators
|
||||
@@ -1901,6 +1876,66 @@ if (!error) {
|
||||
|
||||
This function is particularly useful for extracting data from complex JSON structures with nested arrays and objects. By leveraging wildcards, you can simplify your queries and reduce the need for multiple iterations.
|
||||
|
||||
## C++20 Ranges Support (On-Demand)
|
||||
|
||||
When compiling with C++20 (or later), you can use `std::ranges` with the On-Demand API
|
||||
via the `get_range()` helper. This enables use of range adaptors such as `std::views::transform`.
|
||||
|
||||
```cpp
|
||||
#include "simdjson.h"
|
||||
#include <ranges>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
auto json = R"([
|
||||
{ "name": "Alice", "age": 30 },
|
||||
{ "name": "Bob", "age": 25 },
|
||||
{ "name": "Carol", "age": 35 }
|
||||
])"_padded;
|
||||
|
||||
ondemand::parser parser;
|
||||
auto doc = parser.iterate(json);
|
||||
auto arr = doc.get_array();
|
||||
|
||||
// Use std::views::transform to extract names
|
||||
auto names = ondemand::get_range(arr)
|
||||
| std::views::transform([](auto elem) -> std::string {
|
||||
return std::string(std::string_view(elem["name"]));
|
||||
});
|
||||
|
||||
for (auto name : names) {
|
||||
std::cout << name << std::endl; // Alice, Bob, Carol
|
||||
}
|
||||
```
|
||||
|
||||
The `get_range()` and `get_key_value_range()` functions wrap an `ondemand::array`
|
||||
or `ondemand::object` in a `std::ranges::view` that satisfies `std::ranges::input_range`.
|
||||
They work with both exception and non-exception code:
|
||||
|
||||
```cpp
|
||||
// With exceptions:
|
||||
auto range = ondemand::get_range(doc.get_array());
|
||||
|
||||
// Without exceptions:
|
||||
ondemand::array arr;
|
||||
if (doc.get_array().get(arr) == SUCCESS) {
|
||||
auto range = ondemand::get_range(arr);
|
||||
for (auto elem : range) { /* ... */ }
|
||||
}
|
||||
```
|
||||
|
||||
Object iteration uses `get_key_value_range()` and yields `simdjson_result<ondemand::field>` elements:
|
||||
|
||||
```cpp
|
||||
auto obj = doc.get_object();
|
||||
for (auto field_result : ondemand::get_key_value_range(obj)) {
|
||||
std::cout << field_result.key() << std::endl;
|
||||
}
|
||||
```
|
||||
|
||||
The range wrappers are zero-cost: they forward directly to the underlying
|
||||
On-Demand iterators with no value buffering or extra per-element overhead.
|
||||
|
||||
## Compile-Time JSONPath and JSON Pointer (C++26 Reflection)
|
||||
|
||||
The simdjson library provides **compile-time validated** JSONPath and JSON Pointer accessors when using C++26 Static Reflection. These accessors validate paths against struct definitions at compile time and generate optimized code with zero runtime overhead. In some cases, we find that it is much faster. Furthermore, it is safer in the sense that the expression
|
||||
|
||||
+3
-26
@@ -127,31 +127,11 @@ codepage, and they may call SetFileApisToOEM accordingly.
|
||||
|
||||
|
||||
**Advanced feature:**
|
||||
You can use `simdjson::padded_memory_map` to create a `simdjson::padded_string_view`
|
||||
from a file on disk without copying the file contents into your own buffer.
|
||||
On POSIX systems (Linux, macOS, BSD, ...) it uses `mmap` for true zero-copy
|
||||
access. On Windows it is available as an **opt-in** feature and requires:
|
||||
|
||||
1. Building simdjson with `-DSIMDJSON_ENABLE_MEMORY_FILE_MAPPING_ON_WINDOWS=ON`, or
|
||||
defining `SIMDJSON_ENABLE_MEMORY_FILE_MAPPING_ON_WINDOWS=1` and raising
|
||||
`NTDDI_VERSION` to at least `NTDDI_WIN10_RS4` (Windows 10, version 1803)
|
||||
and linking `onecore.lib` manually if you are consuming simdjson as a
|
||||
pre-built library.
|
||||
2. `#include <windows.h>` before `#include "simdjson.h"` in every
|
||||
translation unit where you want to use `padded_memory_map`.
|
||||
|
||||
When enabled on Windows, the implementation uses `CreateFileMapping2` and
|
||||
`MapViewOfFile3` for true zero-copy mapping whenever the file does not end
|
||||
within `SIMDJSON_PADDING` bytes of a page boundary; otherwise it falls back
|
||||
to reading the file into a padded heap buffer. If those requirements are
|
||||
not met, the class is not declared and the code below will fail to compile.
|
||||
On non-Windows systems, you can use memory-file mapping to create a `simdjson::padded_string_view`
|
||||
from a file on disk.
|
||||
|
||||
```cpp
|
||||
#ifdef _WIN32
|
||||
#include <windows.h> // Must come BEFORE <simdjson.h> on Windows
|
||||
#endif
|
||||
#include "simdjson.h"
|
||||
// ...
|
||||
// if the macro _WIN32 is defined, this will not work since we do not support Windows
|
||||
simdjson::padded_memory_map map(TWITTER_JSON);
|
||||
if (!map.is_valid()) { /* handle error */ }
|
||||
simdjson::padded_string_view view = map.view(); // view is usable while padded_memory_map is in scope
|
||||
@@ -790,9 +770,6 @@ void basics_treewalk_1() {
|
||||
}
|
||||
```
|
||||
|
||||
Notice that we do not include `dom::element_type::BIGINT` in this example
|
||||
as `dom::element_type::BIGINT` type is only generated if the parser was
|
||||
set to support big integers (`parser.number_as_string(true)`).
|
||||
|
||||
|
||||
Reusing the parser for maximum efficiency
|
||||
|
||||
+5
-72
@@ -22,7 +22,6 @@ Contents
|
||||
- [Threads](#threads)
|
||||
- [Support](#support)
|
||||
- [API](#api)
|
||||
- [Streaming directly from a memory-mapped file](#streaming-directly-from-a-memory-mapped-file)
|
||||
- [Use cases](#use-cases)
|
||||
- [Tracking your position](#tracking-your-position)
|
||||
- [Incomplete streams](#incomplete-streams)
|
||||
@@ -157,79 +156,13 @@ for (auto doc : docs) {
|
||||
See [basics.md](basics.md#newline-delimited-json-ndjson-and-json-lines) for an overview of the API.
|
||||
|
||||
|
||||
Streaming directly from a memory-mapped file
|
||||
--------------------------------------------
|
||||
|
||||
When your input is a large NDJSON / JSON-lines file on disk, the most efficient
|
||||
way to feed `iterate_many` is to use `simdjson::padded_memory_map`. It returns
|
||||
a `padded_string_view` with the right amount of trailing padding, so you can
|
||||
hand it straight to `iterate_many` without ever copying the file contents into
|
||||
your own buffer.
|
||||
|
||||
`padded_memory_map` is available on POSIX systems (Linux, macOS, BSD, ...) by
|
||||
default. On Windows it is an **opt-in** feature with the following
|
||||
requirements:
|
||||
|
||||
1. Build simdjson with `-DSIMDJSON_ENABLE_MEMORY_FILE_MAPPING_ON_WINDOWS=ON`, or — if
|
||||
you consume simdjson as a pre-built library — define
|
||||
`SIMDJSON_ENABLE_MEMORY_FILE_MAPPING_ON_WINDOWS=1`, raise `NTDDI_VERSION` to at
|
||||
least `NTDDI_WIN10_RS4` (`0x0A000005`, Windows 10 version 1803), and
|
||||
add `onecore.lib` to your link line yourself. The Windows
|
||||
implementation uses the modern memory APIs `CreateFileMapping2` /
|
||||
`MapViewOfFile3`, which are available starting with that version of
|
||||
Windows and are exported by `onecore.lib`.
|
||||
2. `#include <windows.h>` before `#include "simdjson.h"` in every
|
||||
translation unit where you want to use `padded_memory_map`. simdjson
|
||||
deliberately does not pull in `<windows.h>` itself, so the class is
|
||||
only declared when the Win32 types are already visible.
|
||||
|
||||
If either requirement is not met on Windows, the `padded_memory_map` class is
|
||||
not declared at all and any code that references it fails to compile with an
|
||||
"unknown identifier" error. The availability of the class can be tested with
|
||||
the macro `SIMDJSON_HAS_PADDED_MEMORY_MAP`.
|
||||
|
||||
On POSIX, `padded_memory_map` uses `mmap` to map the file directly into
|
||||
memory with zero copies. On Windows (when enabled), it uses
|
||||
`CreateFileMapping2` + `MapViewOfFile3` for true zero-copy mapping
|
||||
whenever the file does not end within `SIMDJSON_PADDING` bytes of a page
|
||||
boundary; for those rare cases, it transparently falls back to reading
|
||||
the file into a heap-allocated padded buffer so that the returned view
|
||||
always has `SIMDJSON_PADDING` accessible zero bytes after the file content.
|
||||
|
||||
```cpp
|
||||
#ifdef _WIN32
|
||||
#include <windows.h> // Must come BEFORE <simdjson.h> on Windows
|
||||
#endif
|
||||
#include "simdjson.h"
|
||||
|
||||
// ...
|
||||
|
||||
simdjson::padded_memory_map map("huge_stream.ndjson");
|
||||
if (!map.is_valid()) { /* file missing, unreadable, too large, ... */ return; }
|
||||
|
||||
simdjson::ondemand::parser parser;
|
||||
simdjson::ondemand::document_stream stream;
|
||||
auto error = parser.iterate_many(map.view()).get(stream);
|
||||
if (error) { std::cerr << error << std::endl; return; }
|
||||
|
||||
for (auto doc : stream) {
|
||||
// process each JSON document in the stream
|
||||
std::cout << doc << std::endl;
|
||||
}
|
||||
```
|
||||
|
||||
Important lifetime rule: the `padded_string_view` returned by `map.view()` is
|
||||
only valid while the `padded_memory_map` instance is alive, so keep `map`
|
||||
alive for as long as you are iterating the stream.
|
||||
|
||||
The file must not be modified while the memory map is in use. If you need a
|
||||
fully independent copy of the data, use `simdjson::padded_string::load(...)`
|
||||
instead.
|
||||
|
||||
If you prefer single-document parsing on a memory-mapped file, the same
|
||||
pattern applies to `parser.iterate(...)`:
|
||||
**Advanced feature:**
|
||||
On non-Windows systems, you can use memory-file mapping to create a `simdjson::padded_string_view`
|
||||
from a file on disk.
|
||||
|
||||
```cpp
|
||||
// If the macro _WIN32 is defined, this will not work since we do not support memory-file mapping
|
||||
// under Windows at this time.
|
||||
simdjson::padded_memory_map map(myfilename);
|
||||
if (!map.is_valid()) { /* handle error */ }
|
||||
simdjson::padded_string_view view = map.view(); // view is usable while padded_memory_map is in scope
|
||||
|
||||
+6
-73
@@ -18,7 +18,6 @@ Contents
|
||||
- [How it works](#how-it-works)
|
||||
- [Support](#support)
|
||||
- [API](#api)
|
||||
- [Streaming directly from a memory-mapped file](#streaming-directly-from-a-memory-mapped-file)
|
||||
- [Use cases](#use-cases)
|
||||
- [Tracking your position](#tracking-your-position)
|
||||
- [Incomplete streams](#incomplete-streams)
|
||||
@@ -219,83 +218,17 @@ got full document at 29
|
||||
|
||||
|
||||
|
||||
Streaming directly from a memory-mapped file
|
||||
--------------------------------------------
|
||||
|
||||
When your input is a large NDJSON / JSON-lines file on disk, the most
|
||||
efficient way to feed `parse_many` is to use `simdjson::padded_memory_map`.
|
||||
It returns a `padded_string_view` with the right amount of trailing padding,
|
||||
so you can pass it directly to `parse_many` without copying the file content
|
||||
into your own buffer first.
|
||||
|
||||
`padded_memory_map` is available on POSIX systems (Linux, macOS, BSD, ...) by
|
||||
default. On Windows it is an **opt-in** feature with the following
|
||||
requirements:
|
||||
|
||||
1. Build simdjson with `-DSIMDJSON_ENABLE_MEMORY_FILE_MAPPING_ON_WINDOWS=ON`, or — if
|
||||
you consume simdjson as a pre-built library — define
|
||||
`SIMDJSON_ENABLE_MEMORY_FILE_MAPPING_ON_WINDOWS=1`, raise `NTDDI_VERSION` to at
|
||||
least `NTDDI_WIN10_RS4` (`0x0A000005`, Windows 10 version 1803), and
|
||||
add `onecore.lib` to your link line yourself. The Windows
|
||||
implementation uses the modern memory APIs `CreateFileMapping2` /
|
||||
`MapViewOfFile3`, which are available starting with that version of
|
||||
Windows and are exported by `onecore.lib`.
|
||||
2. `#include <windows.h>` before `#include "simdjson.h"` in every
|
||||
translation unit where you want to use `padded_memory_map`. simdjson
|
||||
deliberately does not pull in `<windows.h>` itself, so the class is
|
||||
only declared when the Win32 types are already visible.
|
||||
|
||||
If either requirement is not met on Windows, the `padded_memory_map` class is
|
||||
not declared at all and any code that references it fails to compile with an
|
||||
"unknown identifier" error. The availability of the class can be tested with
|
||||
the macro `SIMDJSON_HAS_PADDED_MEMORY_MAP`.
|
||||
|
||||
On POSIX, `padded_memory_map` uses `mmap` to map the file directly into
|
||||
memory with zero copies. On Windows (when enabled), it uses
|
||||
`CreateFileMapping2` + `MapViewOfFile3` for true zero-copy mapping
|
||||
whenever the file does not end within `SIMDJSON_PADDING` bytes of a page
|
||||
boundary; for those rare cases, it transparently falls back to reading
|
||||
the file into a heap-allocated padded buffer so that the returned view
|
||||
always has `SIMDJSON_PADDING` accessible zero bytes after the file content.
|
||||
|
||||
```cpp
|
||||
#ifdef _WIN32
|
||||
#include <windows.h> // Must come BEFORE <simdjson.h> on Windows
|
||||
#endif
|
||||
#include "simdjson.h"
|
||||
|
||||
// ...
|
||||
|
||||
simdjson::padded_memory_map map("huge_stream.ndjson");
|
||||
if (!map.is_valid()) { /* file missing, unreadable, too large, ... */ return; }
|
||||
|
||||
simdjson::dom::parser parser;
|
||||
simdjson::dom::document_stream stream;
|
||||
auto error = parser.parse_many(map.view()).get(stream);
|
||||
if (error) { std::cerr << error << std::endl; return; }
|
||||
|
||||
for (auto doc : stream) {
|
||||
// process each JSON document in the stream
|
||||
std::cout << doc << std::endl;
|
||||
}
|
||||
```
|
||||
|
||||
Important lifetime rule: the `padded_string_view` returned by `map.view()` is
|
||||
only valid while the `padded_memory_map` instance is alive, so keep `map`
|
||||
alive for as long as you are iterating the stream.
|
||||
|
||||
The file must not be modified while the memory map is in use. If you need a
|
||||
fully independent copy of the data, use `simdjson::padded_string::load(...)`
|
||||
instead.
|
||||
|
||||
If you prefer single-document parsing on a memory-mapped file, the same
|
||||
pattern applies to `parser.parse(...)`:
|
||||
**Advanced feature:**
|
||||
On non-Windows systems, you can use memory-file mapping to create a `simdjson::padded_string_view`
|
||||
from a file on disk.
|
||||
|
||||
```cpp
|
||||
// If the macro _WIN32 is defined, this will not work since we do not support memory-file mapping
|
||||
// under Windows at this time.
|
||||
simdjson::padded_memory_map map(myfilename);
|
||||
if (!map.is_valid()) { /* handle error */ }
|
||||
simdjson::padded_string_view view = map.view(); // view is usable while padded_memory_map is in scope
|
||||
simdjson::dom::element doc = parser.parse(view); // parse the JSON
|
||||
ondemand::document doc = parser.iterate(view); // parse the JSON
|
||||
```
|
||||
|
||||
Incomplete streams
|
||||
|
||||
@@ -59,14 +59,6 @@ concept appendable_containers =
|
||||
details::supports_add<T> || details::supports_append<T> ||
|
||||
details::supports_insert<T>) && !string_view_keyed_map<T>;
|
||||
|
||||
/// Check if T is a key_selector type for efficient JSON field lookup
|
||||
template <typename T>
|
||||
concept key_selector_type = requires(T selector) {
|
||||
{ selector.size() } -> std::same_as<std::size_t>;
|
||||
{ selector.index_of(std::string_view{}) } -> std::same_as<std::size_t>;
|
||||
{ selector.get_key(std::size_t{}) } -> std::same_as<std::string_view>;
|
||||
};
|
||||
|
||||
/// Insert into the container however possible
|
||||
template <appendable_containers T, typename... Args>
|
||||
constexpr decltype(auto) emplace_one(T &vec, Args &&...args) {
|
||||
|
||||
@@ -22,8 +22,6 @@ enum class element_type {
|
||||
STRING = '"', ///< std::string_view
|
||||
BOOL = 't', ///< bool
|
||||
NULL_VALUE = 'n', ///< null
|
||||
/// The BIGINT type is for integers that do not fit in 64 bits. It is only present
|
||||
// if you set parser.number_as_string(true).
|
||||
BIGINT = 'Z' ///< std::string_view: big integer stored as raw digit string
|
||||
};
|
||||
|
||||
|
||||
@@ -181,9 +181,6 @@ inline simdjson_result<document_stream> parser::parse_many(const std::string &s,
|
||||
inline simdjson_result<document_stream> parser::parse_many(const padded_string &s, size_t batch_size) noexcept {
|
||||
return parse_many(s.data(), s.length(), batch_size);
|
||||
}
|
||||
inline simdjson_result<document_stream> parser::parse_many(const padded_string_view &v, size_t batch_size) noexcept {
|
||||
return parse_many(v.data(), v.length(), batch_size);
|
||||
}
|
||||
|
||||
inline simdjson_result<document_stream> parser::parse_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) noexcept {
|
||||
if(batch_size < MINIMAL_BATCH_SIZE) { batch_size = MINIMAL_BATCH_SIZE; }
|
||||
@@ -220,9 +217,6 @@ inline simdjson_result<document_stream> parser::parse_many(const std::string &s,
|
||||
inline simdjson_result<document_stream> parser::parse_many(const padded_string &s, size_t batch_size, stream_format format) noexcept {
|
||||
return parse_many(s.data(), s.length(), batch_size, format);
|
||||
}
|
||||
inline simdjson_result<document_stream> parser::parse_many(const padded_string_view &v, size_t batch_size, stream_format format) noexcept {
|
||||
return parse_many(v.data(), v.length(), batch_size, format);
|
||||
}
|
||||
|
||||
simdjson_inline size_t parser::capacity() const noexcept {
|
||||
return implementation ? implementation->capacity() : 0;
|
||||
|
||||
@@ -490,16 +490,6 @@ public:
|
||||
/** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size) */
|
||||
inline simdjson_result<document_stream> parse_many(const padded_string &s, size_t batch_size = dom::DEFAULT_BATCH_SIZE) noexcept;
|
||||
inline simdjson_result<document_stream> parse_many(const padded_string &&s, size_t batch_size) = delete;// unsafe
|
||||
/** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size)
|
||||
*
|
||||
* Because padded_string_view guarantees SIMDJSON_PADDING trailing bytes, this
|
||||
* overload is safe to use with buffers that the caller owns elsewhere (for
|
||||
* example, a padded_memory_map), with no extra copy. Without this overload,
|
||||
* passing a padded_string_view would silently bind to the padded_string
|
||||
* overload via an implicit conversion, allocating and copying the input, and
|
||||
* — because that temporary is destroyed at the end of the full-expression —
|
||||
* leaving the returned document_stream pointing at freed memory. */
|
||||
inline simdjson_result<document_stream> parse_many(const padded_string_view &v, size_t batch_size = dom::DEFAULT_BATCH_SIZE) noexcept;
|
||||
|
||||
/** @private We do not want to allow implicit conversion from C string to std::string. */
|
||||
simdjson_result<document_stream> parse_many(const char *buf, size_t batch_size = dom::DEFAULT_BATCH_SIZE) noexcept = delete;
|
||||
@@ -520,8 +510,6 @@ public:
|
||||
inline simdjson_result<document_stream> parse_many(const std::string &s, size_t batch_size, stream_format format) noexcept;
|
||||
/** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
|
||||
inline simdjson_result<document_stream> parse_many(const padded_string &s, size_t batch_size, stream_format format) noexcept;
|
||||
/** @overload parse_many(const uint8_t *buf, size_t len, size_t batch_size, stream_format format) */
|
||||
inline simdjson_result<document_stream> parse_many(const padded_string_view &v, size_t batch_size, stream_format format) noexcept;
|
||||
|
||||
/**
|
||||
* Ensure this parser has enough memory to process JSON documents up to `capacity` bytes in length
|
||||
|
||||
@@ -20,9 +20,9 @@
|
||||
#include "simdjson/generic/ondemand/document.h"
|
||||
#include "simdjson/generic/ondemand/document_stream.h"
|
||||
#include "simdjson/generic/ondemand/field.h"
|
||||
#include "simdjson/generic/ondemand/key_selector.h"
|
||||
#include "simdjson/generic/ondemand/object.h"
|
||||
#include "simdjson/generic/ondemand/object_iterator.h"
|
||||
#include "simdjson/generic/ondemand/ranges.h"
|
||||
#include "simdjson/generic/ondemand/serialization.h"
|
||||
|
||||
// Deserialization for standard types
|
||||
@@ -40,6 +40,7 @@
|
||||
#include "simdjson/generic/ondemand/logger-inl.h"
|
||||
#include "simdjson/generic/ondemand/object-inl.h"
|
||||
#include "simdjson/generic/ondemand/object_iterator-inl.h"
|
||||
#include "simdjson/generic/ondemand/ranges-inl.h"
|
||||
#include "simdjson/generic/ondemand/parser-inl.h"
|
||||
#include "simdjson/generic/ondemand/raw_json_string-inl.h"
|
||||
#include "simdjson/generic/ondemand/token_iterator-inl.h"
|
||||
|
||||
@@ -40,6 +40,13 @@ class token_iterator;
|
||||
class value;
|
||||
class value_iterator;
|
||||
|
||||
#if SIMDJSON_SUPPORTS_RANGES
|
||||
class array_range;
|
||||
class array_range_iterator;
|
||||
class object_range;
|
||||
class object_range_iterator;
|
||||
#endif // SIMDJSON_SUPPORTS_RANGES
|
||||
|
||||
} // namespace ondemand
|
||||
} // namespace SIMDJSON_IMPLEMENTATION
|
||||
} // namespace simdjson
|
||||
|
||||
@@ -1,339 +0,0 @@
|
||||
#ifndef SIMDJSON_GENERIC_ONDEMAND_KEY_SELECTOR_H
|
||||
#define SIMDJSON_GENERIC_ONDEMAND_KEY_SELECTOR_H
|
||||
|
||||
#include "simdjson/base.h"
|
||||
#include "simdjson/common_defs.h"
|
||||
#include <array>
|
||||
#include <string_view>
|
||||
#include <cstddef>
|
||||
#include <cstdint>
|
||||
#include <cstring>
|
||||
|
||||
#if SIMDJSON_SUPPORTS_CONCEPTS
|
||||
|
||||
namespace simdjson {
|
||||
namespace SIMDJSON_IMPLEMENTATION {
|
||||
namespace ondemand {
|
||||
|
||||
|
||||
|
||||
// Forward declaration
|
||||
class object;
|
||||
|
||||
/**
|
||||
* A compile-time key selector for efficient JSON object field lookup.
|
||||
* Uses perfect hashing (gperf-style) to map keys to identifiers.
|
||||
*/
|
||||
template <std::size_t N>
|
||||
class key_selector {
|
||||
static_assert(N > 0, "key_selector requires at least one key");
|
||||
static_assert(N <= 100, "key_selector supports at most 100 keys");
|
||||
|
||||
// Perfect hash table data (gperf-style)
|
||||
static constexpr std::size_t MAX_POSITIONS = 16;
|
||||
static constexpr std::size_t POS_LAST_CHAR = std::size_t(-1);
|
||||
static constexpr std::size_t MAX_TABLE_SIZE = 256; // Power of 2, fits in uint8_t
|
||||
|
||||
std::array<std::array<std::uint8_t, 256>, MAX_POSITIONS> asso_values_{};
|
||||
std::uint8_t num_positions_{};
|
||||
std::array<std::size_t, MAX_POSITIONS> positions_{};
|
||||
std::array<std::uint8_t, MAX_TABLE_SIZE> slot_to_key_{};
|
||||
std::array<std::uint8_t, N> key_to_slot_{};
|
||||
std::array<std::array<char, 64>, N> key_data_{};
|
||||
std::array<std::uint8_t, N> key_lengths_{};
|
||||
std::size_t table_size_{};
|
||||
|
||||
public:
|
||||
// Validate keys at compile time
|
||||
constexpr void validate_keys(const std::array<std::string_view, N>& keys) {
|
||||
for (std::size_t i = 0; i < N; ++i) {
|
||||
auto key = keys[i];
|
||||
if (key.empty()) {
|
||||
throw "Empty keys are not allowed in key_selector";
|
||||
}
|
||||
if (key.size() > SIMDJSON_PADDING) {
|
||||
throw "Key length exceeds SIMDJSON_PADDING (64 bytes)";
|
||||
}
|
||||
for (char c : key) {
|
||||
if (c == '\\') {
|
||||
throw "Escape characters (\\) are not allowed in key_selector keys";
|
||||
}
|
||||
if (c == '\0') {
|
||||
throw "Null characters are not allowed in key_selector keys";
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Gperf-style perfect hash generation using partition-based algorithm
|
||||
constexpr void generate_hash_table(const std::array<std::string_view, N>& keys) {
|
||||
// Try power-of-two table sizes starting from next_power_of_2(N)
|
||||
constexpr std::size_t START_M = next_power_of_2(N);
|
||||
if constexpr (START_M <= MAX_TABLE_SIZE) {
|
||||
if (try_compute_phf<START_M>(keys)) return;
|
||||
if constexpr (START_M * 2 <= MAX_TABLE_SIZE) {
|
||||
if (try_compute_phf<START_M * 2>(keys)) return;
|
||||
if constexpr (START_M * 4 <= MAX_TABLE_SIZE) {
|
||||
if (try_compute_phf<START_M * 4>(keys)) return;
|
||||
if constexpr (START_M * 8 <= MAX_TABLE_SIZE) {
|
||||
if (try_compute_phf<START_M * 8>(keys)) return;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Fallback: linear table
|
||||
table_size_ = N;
|
||||
num_positions_ = 0;
|
||||
std::fill(slot_to_key_.begin(), slot_to_key_.begin() + MAX_TABLE_SIZE, static_cast<std::uint8_t>(N));
|
||||
for (std::size_t i = 0; i < N; ++i) {
|
||||
slot_to_key_[i] = static_cast<std::uint8_t>(i);
|
||||
key_to_slot_[i] = static_cast<std::uint8_t>(i);
|
||||
}
|
||||
}
|
||||
|
||||
private:
|
||||
// Helper functions for gperf algorithm
|
||||
static constexpr std::size_t next_power_of_2(std::size_t n) {
|
||||
if (n == 0) return 1;
|
||||
std::size_t p = 1;
|
||||
while (p < n) p <<= 1;
|
||||
return p;
|
||||
}
|
||||
|
||||
static constexpr std::size_t char_at(std::string_view key, std::size_t pos) {
|
||||
if (pos == POS_LAST_CHAR) {
|
||||
return key.empty() ? 256 : static_cast<unsigned char>(key.back());
|
||||
}
|
||||
return (pos < key.size()) ? static_cast<unsigned char>(key[pos]) : 256;
|
||||
}
|
||||
|
||||
template <std::size_t M>
|
||||
constexpr bool try_compute_phf(const std::array<std::string_view, N>& keys) {
|
||||
// Initialize
|
||||
std::array<std::array<std::size_t, 256>, MAX_POSITIONS> asso{};
|
||||
std::size_t npos = 0;
|
||||
std::array<std::size_t, MAX_POSITIONS> pos{};
|
||||
std::array<std::size_t, M> s2k{};
|
||||
|
||||
// Try to generate gperf
|
||||
if (try_generate_gperf<M>(keys, asso, npos, pos, s2k)) {
|
||||
table_size_ = M;
|
||||
num_positions_ = static_cast<std::uint8_t>(npos);
|
||||
for (std::size_t p = 0; p < MAX_POSITIONS; ++p) {
|
||||
positions_[p] = pos[p];
|
||||
for (std::size_t c = 0; c < 256; ++c) {
|
||||
asso_values_[p][c] = static_cast<std::uint8_t>(asso[p][c]);
|
||||
}
|
||||
}
|
||||
for (std::size_t i = 0; i < M; ++i) {
|
||||
slot_to_key_[i] = static_cast<std::uint8_t>(s2k[i]);
|
||||
}
|
||||
// Fill remaining slots with sentinel
|
||||
for (std::size_t i = M; i < MAX_TABLE_SIZE; ++i) {
|
||||
slot_to_key_[i] = static_cast<std::uint8_t>(N);
|
||||
}
|
||||
|
||||
// Build key_to_slot mapping
|
||||
for (std::size_t i = 0; i < N; ++i) {
|
||||
key_to_slot_[i] = static_cast<std::uint8_t>(N); // Initialize
|
||||
}
|
||||
for (std::size_t slot = 0; slot < M; ++slot) {
|
||||
std::size_t key_idx = s2k[slot];
|
||||
if (key_idx < N) {
|
||||
key_to_slot_[key_idx] = static_cast<std::uint8_t>(slot);
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
template <std::size_t M>
|
||||
static constexpr bool try_generate_gperf(
|
||||
const std::array<std::string_view, N>& keys,
|
||||
std::array<std::array<std::size_t, 256>, MAX_POSITIONS>& asso_values,
|
||||
std::size_t& num_positions,
|
||||
std::array<std::size_t, MAX_POSITIONS>& positions,
|
||||
std::array<std::size_t, M>& slot_to_key)
|
||||
{
|
||||
// Initialize
|
||||
for (std::size_t p = 0; p < MAX_POSITIONS; ++p) {
|
||||
for (std::size_t c = 0; c < 256; ++c) {
|
||||
asso_values[p][c] = 0;
|
||||
}
|
||||
}
|
||||
for (std::size_t i = 0; i < M; ++i) {
|
||||
slot_to_key[i] = N;
|
||||
}
|
||||
|
||||
// Try length-only hashing first
|
||||
bool success = true;
|
||||
for (std::size_t i = 0; i < N && success; ++i) {
|
||||
std::size_t slot = keys[i].size() % M;
|
||||
if (slot_to_key[slot] != N) {
|
||||
success = false;
|
||||
} else {
|
||||
slot_to_key[slot] = i;
|
||||
}
|
||||
}
|
||||
|
||||
if (success) {
|
||||
num_positions = 0;
|
||||
return true;
|
||||
}
|
||||
|
||||
// Try with position 0
|
||||
positions[0] = 0;
|
||||
num_positions = 1;
|
||||
|
||||
// Find a working assignment of asso_values for position 0
|
||||
// Use a simple approach: try different offsets
|
||||
for (std::size_t offset = 0; offset < M; ++offset) {
|
||||
// Reset
|
||||
for (std::size_t i = 0; i < M; ++i) {
|
||||
slot_to_key[i] = N;
|
||||
}
|
||||
|
||||
// Assign asso_values based on offset
|
||||
for (std::size_t c = 0; c < 256; ++c) {
|
||||
asso_values[0][c] = (c + offset) % M;
|
||||
}
|
||||
|
||||
success = true;
|
||||
for (std::size_t i = 0; i < N && success; ++i) {
|
||||
std::size_t h = keys[i].size();
|
||||
std::size_t ch = char_at(keys[i], 0);
|
||||
if (ch < 256) h += asso_values[0][ch];
|
||||
std::size_t slot = h % M;
|
||||
|
||||
if (slot_to_key[slot] != N) {
|
||||
success = false;
|
||||
} else {
|
||||
slot_to_key[slot] = i;
|
||||
}
|
||||
}
|
||||
|
||||
if (success) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
// Try with positions {0, last_char}
|
||||
if (N <= 50) { // Only for smaller N to avoid complexity
|
||||
positions[0] = 0;
|
||||
positions[1] = POS_LAST_CHAR;
|
||||
num_positions = 2;
|
||||
|
||||
for (std::size_t offset1 = 0; offset1 < 4 && !success; ++offset1) {
|
||||
for (std::size_t offset2 = 0; offset2 < 4 && !success; ++offset2) {
|
||||
// Reset
|
||||
for (std::size_t i = 0; i < M; ++i) {
|
||||
slot_to_key[i] = N;
|
||||
}
|
||||
|
||||
// Assign asso_values
|
||||
for (std::size_t c = 0; c < 256; ++c) {
|
||||
asso_values[0][c] = (c + offset1) % M;
|
||||
asso_values[1][c] = (c + offset2) % M;
|
||||
}
|
||||
|
||||
success = true;
|
||||
for (std::size_t i = 0; i < N && success; ++i) {
|
||||
std::size_t h = keys[i].size();
|
||||
std::size_t ch1 = char_at(keys[i], 0);
|
||||
if (ch1 < 256) h += asso_values[0][ch1];
|
||||
std::size_t ch2 = char_at(keys[i], POS_LAST_CHAR);
|
||||
if (ch2 < 256) h += asso_values[1][ch2];
|
||||
std::size_t slot = h % M;
|
||||
|
||||
if (slot_to_key[slot] != N) {
|
||||
success = false;
|
||||
} else {
|
||||
slot_to_key[slot] = i;
|
||||
}
|
||||
}
|
||||
|
||||
if (success) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
|
||||
|
||||
public:
|
||||
constexpr key_selector(const std::array<std::string_view, N>& keys) {
|
||||
validate_keys(keys);
|
||||
|
||||
// Store key data
|
||||
for (std::size_t i = 0; i < N; ++i) {
|
||||
key_lengths_[i] = static_cast<std::uint8_t>(keys[i].size());
|
||||
std::copy(keys[i].begin(), keys[i].end(), key_data_[i].begin());
|
||||
}
|
||||
|
||||
generate_hash_table(keys);
|
||||
}
|
||||
|
||||
[[nodiscard]] constexpr std::size_t size() const noexcept { return N; }
|
||||
|
||||
[[nodiscard]] constexpr simdjson_really_inline std::size_t compute_hash(std::string_view key) const noexcept {
|
||||
std::size_t h = key.size();
|
||||
const char* kp = key.data();
|
||||
for (std::uint8_t i = 0; i < num_positions_; ++i) {
|
||||
std::size_t pos = positions_[i];
|
||||
std::size_t ch;
|
||||
if (pos == POS_LAST_CHAR) {
|
||||
ch = static_cast<unsigned char>(key.back());
|
||||
} else {
|
||||
ch = static_cast<unsigned char>(kp[pos]);
|
||||
}
|
||||
h += asso_values_[i][ch];
|
||||
}
|
||||
return h & (table_size_ - 1);
|
||||
}
|
||||
|
||||
[[nodiscard]] constexpr simdjson_really_inline bool contains(std::string_view key) const noexcept {
|
||||
std::size_t slot = compute_hash(key);
|
||||
if (slot >= table_size_) return false;
|
||||
|
||||
std::uint8_t key_idx = slot_to_key_[slot];
|
||||
if (key_idx >= N) return false;
|
||||
|
||||
// Compare key
|
||||
if (key_lengths_[key_idx] != key.size()) return false;
|
||||
return std::equal(key.begin(), key.end(), key_data_[key_idx].begin());
|
||||
}
|
||||
|
||||
[[nodiscard]] constexpr simdjson_really_inline std::size_t index_of(std::string_view key) const noexcept {
|
||||
std::size_t slot = compute_hash(key);
|
||||
if (slot >= table_size_) return N; // Invalid index
|
||||
|
||||
std::uint8_t key_idx = slot_to_key_[slot];
|
||||
if (key_idx >= N) return N;
|
||||
|
||||
// Compare key
|
||||
if (key_lengths_[key_idx] != key.size()) return N;
|
||||
if (!std::equal(key.begin(), key.end(), key_data_[key_idx].begin())) return N;
|
||||
|
||||
return key_idx;
|
||||
}
|
||||
|
||||
// Accessors for key data (used by object::find_field)
|
||||
[[nodiscard]] constexpr std::string_view get_key(std::size_t index) const noexcept {
|
||||
if (index >= N) return {};
|
||||
return std::string_view(key_data_[index].data(), key_lengths_[index]);
|
||||
}
|
||||
};
|
||||
|
||||
} // namespace ondemand
|
||||
} // namespace SIMDJSON_IMPLEMENTATION
|
||||
} // namespace simdjson
|
||||
|
||||
#endif // SIMDJSON_SUPPORTS_CONCEPTS
|
||||
|
||||
#endif // SIMDJSON_GENERIC_ONDEMAND_KEY_SELECTOR_H
|
||||
@@ -63,42 +63,6 @@ simdjson_inline simdjson_result<value> object::find_field(const std::string_view
|
||||
return value(iter.child());
|
||||
}
|
||||
|
||||
#if SIMDJSON_SUPPORTS_CONCEPTS
|
||||
template <concepts::key_selector_type Selector>
|
||||
simdjson_inline std::pair<std::size_t, simdjson_result<value>> object::find_field(const Selector& selector) & noexcept {
|
||||
// Try to find any of the keys in the selector
|
||||
for (std::size_t i = 0; i < selector.size(); ++i) {
|
||||
std::string_view key = selector.get_key(i);
|
||||
auto result = iter.find_field_unordered_raw(key);
|
||||
if (result.error()) {
|
||||
return {selector.size(), result.error()};
|
||||
}
|
||||
bool has_value = result.value();
|
||||
if (has_value) {
|
||||
return {i, value(iter.child())};
|
||||
}
|
||||
}
|
||||
return {selector.size(), NO_SUCH_FIELD}; // Return size() as invalid index
|
||||
}
|
||||
|
||||
template <concepts::key_selector_type Selector>
|
||||
simdjson_inline std::pair<std::size_t, simdjson_result<value>> object::find_field(const Selector& selector) && noexcept {
|
||||
// Try to find any of the keys in the selector
|
||||
for (std::size_t i = 0; i < selector.size(); ++i) {
|
||||
std::string_view key = selector.get_key(i);
|
||||
auto result = iter.find_field_unordered_raw(key);
|
||||
if (result.error()) {
|
||||
return {selector.size(), result.error()};
|
||||
}
|
||||
bool has_value = result.value();
|
||||
if (has_value) {
|
||||
return {i, value(iter.child())};
|
||||
}
|
||||
}
|
||||
return {selector.size(), NO_SUCH_FIELD}; // Return size() as invalid index
|
||||
}
|
||||
#endif
|
||||
|
||||
simdjson_inline simdjson_result<object> object::start(value_iterator &iter) noexcept {
|
||||
SIMDJSON_TRY( iter.start_object().error() );
|
||||
return object(iter);
|
||||
@@ -370,20 +334,6 @@ simdjson_inline simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::value> simdjs
|
||||
return std::forward<SIMDJSON_IMPLEMENTATION::ondemand::object>(first).find_field(key);
|
||||
}
|
||||
|
||||
#if SIMDJSON_SUPPORTS_CONCEPTS
|
||||
template <concepts::key_selector_type Selector>
|
||||
simdjson_inline std::pair<std::size_t, simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::value>> simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::object>::find_field(const Selector& selector) & noexcept {
|
||||
if (error()) { return {0, error()}; }
|
||||
return first.find_field(selector);
|
||||
}
|
||||
|
||||
template <concepts::key_selector_type Selector>
|
||||
simdjson_inline std::pair<std::size_t, simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::value>> simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::object>::find_field(const Selector& selector) && noexcept {
|
||||
if (error()) { return {0, error()}; }
|
||||
return std::forward<SIMDJSON_IMPLEMENTATION::ondemand::object>(first).find_field(selector);
|
||||
}
|
||||
#endif
|
||||
|
||||
simdjson_inline simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::value> simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::object>::at_pointer(std::string_view json_pointer) noexcept {
|
||||
if (error()) { return error(); }
|
||||
return first.at_pointer(json_pointer);
|
||||
|
||||
@@ -5,7 +5,6 @@
|
||||
#include "simdjson/generic/ondemand/base.h"
|
||||
#include "simdjson/generic/implementation_simdjson_result_base.h"
|
||||
#include "simdjson/generic/ondemand/value_iterator.h"
|
||||
#include "simdjson/generic/ondemand/key_selector.h"
|
||||
#include <vector>
|
||||
#if SIMDJSON_STATIC_REFLECTION && SIMDJSON_SUPPORTS_CONCEPTS
|
||||
#include "simdjson/generic/ondemand/json_string_builder.h" // for constevalutil::fixed_string
|
||||
@@ -123,23 +122,6 @@ public:
|
||||
/** @overload simdjson_inline simdjson_result<value> find_field_unordered(std::string_view key) & noexcept; */
|
||||
simdjson_inline simdjson_result<value> operator[](std::string_view key) && noexcept;
|
||||
|
||||
#if SIMDJSON_SUPPORTS_CONCEPTS
|
||||
/**
|
||||
* Look up a field by name using a key_selector. This method is similar to find_field_unordered()
|
||||
* but uses a compile-time generated perfect hash table for efficient lookup.
|
||||
*
|
||||
* @tparam Selector The key_selector type
|
||||
* @param selector The key selector instance
|
||||
* @returns A pair containing the key identifier (index in the selector) and the value,
|
||||
* or NO_SUCH_FIELD if the field is not in the object.
|
||||
*/
|
||||
template <concepts::key_selector_type Selector>
|
||||
simdjson_inline std::pair<std::size_t, simdjson_result<value>> find_field(const Selector& selector) & noexcept;
|
||||
/** @overload template <concepts::key_selector_type Selector> simdjson_inline std::pair<std::size_t, simdjson_result<value>> find_field(const Selector& selector) & noexcept; */
|
||||
template <concepts::key_selector_type Selector>
|
||||
simdjson_inline std::pair<std::size_t, simdjson_result<value>> find_field(const Selector& selector) && noexcept;
|
||||
#endif
|
||||
|
||||
/**
|
||||
* Get the value associated with the given JSON pointer. We use the RFC 6901
|
||||
* https://tools.ietf.org/html/rfc6901 standard, interpreting the current node
|
||||
@@ -342,12 +324,6 @@ public:
|
||||
simdjson_inline simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::value> find_field(std::string_view key) && noexcept;
|
||||
simdjson_inline simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::value> find_field_unordered(std::string_view key) & noexcept;
|
||||
simdjson_inline simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::value> find_field_unordered(std::string_view key) && noexcept;
|
||||
#if SIMDJSON_SUPPORTS_CONCEPTS
|
||||
template <concepts::key_selector_type Selector>
|
||||
simdjson_inline std::pair<std::size_t, simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::value>> find_field(const Selector& selector) & noexcept;
|
||||
template <concepts::key_selector_type Selector>
|
||||
simdjson_inline std::pair<std::size_t, simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::value>> find_field(const Selector& selector) && noexcept;
|
||||
#endif
|
||||
simdjson_inline simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::value> operator[](std::string_view key) & noexcept;
|
||||
simdjson_inline simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::value> operator[](std::string_view key) && noexcept;
|
||||
simdjson_inline simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::value> at_pointer(std::string_view json_pointer) noexcept;
|
||||
|
||||
@@ -0,0 +1,138 @@
|
||||
#ifndef SIMDJSON_GENERIC_ONDEMAND_RANGES_INL_H
|
||||
|
||||
#ifndef SIMDJSON_CONDITIONAL_INCLUDE
|
||||
#define SIMDJSON_GENERIC_ONDEMAND_RANGES_INL_H
|
||||
#include "simdjson/generic/ondemand/base.h"
|
||||
#include "simdjson/generic/ondemand/ranges.h"
|
||||
#include "simdjson/generic/ondemand/array-inl.h"
|
||||
#include "simdjson/generic/ondemand/array_iterator-inl.h"
|
||||
#include "simdjson/generic/ondemand/object-inl.h"
|
||||
#include "simdjson/generic/ondemand/object_iterator-inl.h"
|
||||
#endif // SIMDJSON_CONDITIONAL_INCLUDE
|
||||
|
||||
#if SIMDJSON_SUPPORTS_RANGES
|
||||
|
||||
namespace simdjson {
|
||||
namespace SIMDJSON_IMPLEMENTATION {
|
||||
namespace ondemand {
|
||||
|
||||
//
|
||||
// array_range_iterator
|
||||
//
|
||||
|
||||
simdjson_inline array_range_iterator::array_range_iterator(array_iterator iter) noexcept
|
||||
: iter_{iter} {}
|
||||
|
||||
simdjson_inline simdjson_result<value> array_range_iterator::operator*() const noexcept {
|
||||
return *iter_;
|
||||
}
|
||||
|
||||
simdjson_inline array_range_iterator& array_range_iterator::operator++() noexcept {
|
||||
++iter_;
|
||||
return *this;
|
||||
}
|
||||
|
||||
SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
|
||||
simdjson_inline void array_range_iterator::operator++(int) noexcept {
|
||||
++*this;
|
||||
}
|
||||
SIMDJSON_POP_DISABLE_WARNINGS
|
||||
|
||||
//
|
||||
// array_range
|
||||
//
|
||||
|
||||
simdjson_inline array_range::array_range(array& arr) noexcept {
|
||||
auto b = arr.begin();
|
||||
if (b.error()) { error_ = b.error(); return; }
|
||||
begin_ = b.value_unsafe();
|
||||
end_ = arr.end().value_unsafe();
|
||||
}
|
||||
|
||||
simdjson_inline array_range_iterator array_range::begin() noexcept {
|
||||
return array_range_iterator(begin_);
|
||||
}
|
||||
|
||||
simdjson_inline array_range_iterator array_range::end() noexcept {
|
||||
return array_range_iterator(end_);
|
||||
}
|
||||
|
||||
//
|
||||
// object_range_iterator
|
||||
//
|
||||
|
||||
simdjson_inline object_range_iterator::object_range_iterator(object_iterator iter) noexcept
|
||||
: iter_{iter} {}
|
||||
|
||||
simdjson_inline simdjson_result<field> object_range_iterator::operator*() const noexcept {
|
||||
return *iter_;
|
||||
}
|
||||
|
||||
simdjson_inline object_range_iterator& object_range_iterator::operator++() noexcept {
|
||||
++iter_;
|
||||
return *this;
|
||||
}
|
||||
|
||||
SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
|
||||
simdjson_inline void object_range_iterator::operator++(int) noexcept {
|
||||
++*this;
|
||||
}
|
||||
SIMDJSON_POP_DISABLE_WARNINGS
|
||||
|
||||
|
||||
//
|
||||
// object_range
|
||||
//
|
||||
|
||||
simdjson_inline object_range::object_range(object& obj) noexcept {
|
||||
auto b = obj.begin();
|
||||
if (b.error()) { error_ = b.error(); return; }
|
||||
begin_ = b.value_unsafe();
|
||||
end_ = obj.end().value_unsafe();
|
||||
}
|
||||
|
||||
simdjson_inline object_range_iterator object_range::begin() noexcept {
|
||||
return object_range_iterator(begin_);
|
||||
}
|
||||
|
||||
simdjson_inline object_range_iterator object_range::end() noexcept {
|
||||
return object_range_iterator(end_);
|
||||
}
|
||||
|
||||
//
|
||||
// Free functions
|
||||
//
|
||||
|
||||
simdjson_inline array_range get_range(array& arr) noexcept {
|
||||
return array_range(arr);
|
||||
}
|
||||
|
||||
simdjson_inline object_range get_key_value_range(object& obj) noexcept {
|
||||
return object_range(obj);
|
||||
}
|
||||
|
||||
#if SIMDJSON_EXCEPTIONS
|
||||
simdjson_inline array_range get_range(simdjson_result<array> result) {
|
||||
return array_range(result.value());
|
||||
}
|
||||
|
||||
simdjson_inline object_range get_key_value_range(simdjson_result<object> result) {
|
||||
return object_range(result.value());
|
||||
}
|
||||
#endif // SIMDJSON_EXCEPTIONS
|
||||
|
||||
} // namespace ondemand
|
||||
} // namespace SIMDJSON_IMPLEMENTATION
|
||||
} // namespace simdjson
|
||||
|
||||
// Verify the range wrapper types satisfy the expected C++20 concepts.
|
||||
static_assert(std::input_iterator<simdjson::SIMDJSON_IMPLEMENTATION::ondemand::array_range_iterator>);
|
||||
static_assert(std::input_iterator<simdjson::SIMDJSON_IMPLEMENTATION::ondemand::object_range_iterator>);
|
||||
static_assert(std::ranges::input_range<simdjson::SIMDJSON_IMPLEMENTATION::ondemand::array_range>);
|
||||
static_assert(std::ranges::input_range<simdjson::SIMDJSON_IMPLEMENTATION::ondemand::object_range>);
|
||||
static_assert(std::ranges::view<simdjson::SIMDJSON_IMPLEMENTATION::ondemand::array_range>);
|
||||
static_assert(std::ranges::view<simdjson::SIMDJSON_IMPLEMENTATION::ondemand::object_range>);
|
||||
|
||||
#endif // SIMDJSON_SUPPORTS_RANGES
|
||||
|
||||
#endif // SIMDJSON_GENERIC_ONDEMAND_RANGES_INL_H
|
||||
@@ -0,0 +1,182 @@
|
||||
#ifndef SIMDJSON_GENERIC_ONDEMAND_RANGES_H
|
||||
|
||||
#ifndef SIMDJSON_CONDITIONAL_INCLUDE
|
||||
#define SIMDJSON_GENERIC_ONDEMAND_RANGES_H
|
||||
#include "simdjson/generic/ondemand/base.h"
|
||||
#include "simdjson/generic/ondemand/array.h"
|
||||
#include "simdjson/generic/ondemand/array_iterator.h"
|
||||
#include "simdjson/generic/ondemand/object.h"
|
||||
#include "simdjson/generic/ondemand/object_iterator.h"
|
||||
#include "simdjson/generic/ondemand/field.h"
|
||||
#include "simdjson/generic/ondemand/value.h"
|
||||
#endif // SIMDJSON_CONDITIONAL_INCLUDE
|
||||
|
||||
#if SIMDJSON_SUPPORTS_RANGES
|
||||
|
||||
namespace simdjson {
|
||||
namespace SIMDJSON_IMPLEMENTATION {
|
||||
namespace ondemand {
|
||||
|
||||
/**
|
||||
* A ranges-compatible iterator adapter for JSON arrays.
|
||||
*
|
||||
* Wraps array_iterator to satisfy std::input_iterator by providing:
|
||||
* - const operator* (via mutable internal state)
|
||||
* - post-increment operator
|
||||
* - iterator_concept tag
|
||||
*
|
||||
* The mutable approach is standard for single-pass input iterators that
|
||||
* read from external sources (similar to std::istream_iterator).
|
||||
*/
|
||||
class array_range_iterator {
|
||||
public:
|
||||
using iterator_concept = std::input_iterator_tag;
|
||||
using value_type = simdjson_result<value>;
|
||||
using reference = simdjson_result<value>;
|
||||
using difference_type = std::ptrdiff_t;
|
||||
|
||||
simdjson_inline array_range_iterator() noexcept = default;
|
||||
simdjson_inline explicit array_range_iterator(array_iterator iter) noexcept;
|
||||
|
||||
/**
|
||||
* Get the current element. Const-qualified for std::indirectly_readable;
|
||||
* internally delegates to the mutable wrapped iterator.
|
||||
*/
|
||||
simdjson_inline simdjson_result<value> operator*() const noexcept;
|
||||
simdjson_inline array_range_iterator& operator++() noexcept;
|
||||
SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
|
||||
simdjson_inline void operator++(int) noexcept;
|
||||
SIMDJSON_POP_DISABLE_WARNINGS
|
||||
/**
|
||||
* Comparison delegates to array_iterator::operator==, which checks
|
||||
* whether the underlying parser has finished the array (depth-based).
|
||||
*/
|
||||
simdjson_inline friend bool operator==(const array_range_iterator& a,
|
||||
const array_range_iterator& b) noexcept {
|
||||
return a.iter_ == b.iter_;
|
||||
}
|
||||
|
||||
private:
|
||||
mutable array_iterator iter_{};
|
||||
};
|
||||
|
||||
/**
|
||||
* A std::ranges::view over a JSON array.
|
||||
*
|
||||
* Wraps an ondemand::array and exposes begin()/end() that return
|
||||
* array_range_iterator (satisfying std::input_iterator), enabling
|
||||
* use with std::views::transform and other range adaptors.
|
||||
*
|
||||
* If the array's begin() returns an error (only possible under
|
||||
* SIMDJSON_DEVELOPMENT_CHECKS), the range will be empty and error()
|
||||
* will return the error code.
|
||||
*
|
||||
* Usage:
|
||||
* ondemand::parser parser;
|
||||
* auto doc = parser.iterate(json);
|
||||
* auto arr = doc.get_array().value();
|
||||
* for (auto elem : ondemand::get_range(arr)) { ... }
|
||||
*/
|
||||
class array_range {
|
||||
public:
|
||||
simdjson_inline array_range() noexcept = default;
|
||||
simdjson_inline explicit array_range(array& arr) noexcept;
|
||||
|
||||
simdjson_inline array_range_iterator begin() noexcept;
|
||||
simdjson_inline array_range_iterator end() noexcept;
|
||||
|
||||
/** Returns SUCCESS if the range was created successfully, or the error code otherwise. */
|
||||
simdjson_inline error_code error() const noexcept { return error_; }
|
||||
|
||||
private:
|
||||
array_iterator begin_{};
|
||||
array_iterator end_{};
|
||||
error_code error_{SUCCESS};
|
||||
};
|
||||
|
||||
/**
|
||||
* A ranges-compatible iterator adapter for JSON objects.
|
||||
*
|
||||
* Wraps object_iterator to satisfy std::input_iterator, yielding
|
||||
* simdjson_result<field> elements (key-value pairs).
|
||||
*/
|
||||
class object_range_iterator {
|
||||
public:
|
||||
using iterator_concept = std::input_iterator_tag;
|
||||
using value_type = simdjson_result<field>;
|
||||
using reference = simdjson_result<field>;
|
||||
using difference_type = std::ptrdiff_t;
|
||||
|
||||
simdjson_inline object_range_iterator() noexcept = default;
|
||||
simdjson_inline explicit object_range_iterator(object_iterator iter) noexcept;
|
||||
|
||||
simdjson_inline simdjson_result<field> operator*() const noexcept;
|
||||
simdjson_inline object_range_iterator& operator++() noexcept;
|
||||
SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
|
||||
simdjson_inline void operator++(int) noexcept;
|
||||
SIMDJSON_POP_DISABLE_WARNINGS
|
||||
simdjson_inline friend bool operator==(const object_range_iterator& a,
|
||||
const object_range_iterator& b) noexcept {
|
||||
return a.iter_ == b.iter_;
|
||||
}
|
||||
|
||||
private:
|
||||
mutable object_iterator iter_{};
|
||||
};
|
||||
|
||||
/**
|
||||
* A std::ranges::view over a JSON object.
|
||||
*
|
||||
* Wraps an ondemand::object and exposes begin()/end() that return
|
||||
* object_range_iterator, enabling use with range adaptors.
|
||||
*
|
||||
* If the object's begin() returns an error, the range will be empty
|
||||
* and error() will return the error code.
|
||||
*/
|
||||
class object_range {
|
||||
public:
|
||||
simdjson_inline object_range() noexcept = default;
|
||||
simdjson_inline explicit object_range(object& obj) noexcept;
|
||||
|
||||
simdjson_inline object_range_iterator begin() noexcept;
|
||||
simdjson_inline object_range_iterator end() noexcept;
|
||||
|
||||
/** Returns SUCCESS if the range was created successfully, or the error code otherwise. */
|
||||
simdjson_inline error_code error() const noexcept { return error_; }
|
||||
|
||||
private:
|
||||
object_iterator begin_{};
|
||||
object_iterator end_{};
|
||||
error_code error_{SUCCESS};
|
||||
};
|
||||
|
||||
/** Get a std::ranges compatible view over a JSON array. */
|
||||
simdjson_inline array_range get_range(array& arr) noexcept;
|
||||
|
||||
/** Get a std::ranges compatible view over a JSON object (key-value pairs). */
|
||||
simdjson_inline object_range get_key_value_range(object& obj) noexcept;
|
||||
|
||||
#if SIMDJSON_EXCEPTIONS
|
||||
/** Get a std::ranges compatible view, unwrapping the simdjson_result (throws on error). */
|
||||
simdjson_inline array_range get_range(simdjson_result<array> result);
|
||||
|
||||
/** Get a std::ranges compatible view, unwrapping the simdjson_result (throws on error). */
|
||||
simdjson_inline object_range get_key_value_range(simdjson_result<object> result);
|
||||
#endif // SIMDJSON_EXCEPTIONS
|
||||
|
||||
} // namespace ondemand
|
||||
} // namespace SIMDJSON_IMPLEMENTATION
|
||||
} // namespace simdjson
|
||||
|
||||
namespace std {
|
||||
namespace ranges {
|
||||
template<>
|
||||
inline constexpr bool enable_view<simdjson::SIMDJSON_IMPLEMENTATION::ondemand::array_range> = true;
|
||||
template<>
|
||||
inline constexpr bool enable_view<simdjson::SIMDJSON_IMPLEMENTATION::ondemand::object_range> = true;
|
||||
} // namespace ranges
|
||||
} // namespace std
|
||||
|
||||
#endif // SIMDJSON_SUPPORTS_RANGES
|
||||
|
||||
#endif // SIMDJSON_GENERIC_ONDEMAND_RANGES_H
|
||||
@@ -10,18 +10,13 @@
|
||||
#include <climits>
|
||||
#include <cwchar>
|
||||
|
||||
#if SIMDJSON_HAS_UNISTD_H
|
||||
#ifndef _WIN32
|
||||
#include <fcntl.h>
|
||||
#include <stdio.h>
|
||||
#include <sys/mman.h>
|
||||
#include <sys/stat.h>
|
||||
#include <unistd.h>
|
||||
#endif
|
||||
// On Windows, `padded_memory_map` (when it is enabled) depends on types and
|
||||
// functions declared in <windows.h>. We deliberately do NOT include that
|
||||
// header here: users of simdjson who want `padded_memory_map` on Windows
|
||||
// must include <windows.h> themselves *before* including this header. See
|
||||
// padded_string.h for the detection logic.
|
||||
|
||||
namespace simdjson {
|
||||
namespace internal {
|
||||
@@ -390,9 +385,7 @@ inline bool padded_string_builder::reserve(size_t additional) noexcept {
|
||||
}
|
||||
|
||||
|
||||
#if SIMDJSON_HAS_PADDED_MEMORY_MAP
|
||||
|
||||
#if SIMDJSON_HAS_UNISTD_H
|
||||
#ifndef _WIN32
|
||||
simdjson_inline padded_memory_map::padded_memory_map(const char *filename) noexcept {
|
||||
|
||||
int fd = open(filename, O_RDONLY);
|
||||
@@ -428,132 +421,7 @@ simdjson_inline padded_memory_map::~padded_memory_map() noexcept {
|
||||
munmap(const_cast<char *>(data), size + simdjson::SIMDJSON_PADDING);
|
||||
}
|
||||
}
|
||||
#elif defined(_WIN32)
|
||||
// Windows zero-copy implementation using placeholder virtual memory.
|
||||
//
|
||||
// We use the modern Windows memory APIs (VirtualAlloc2, CreateFileMapping2,
|
||||
// MapViewOfFile3 — available since Windows 10 1803) to map the file into a
|
||||
// contiguous virtual address range that includes at least SIMDJSON_PADDING
|
||||
// zero bytes after the file content, with no data copies.
|
||||
//
|
||||
// Strategy:
|
||||
// 1. If rounding the file size up to the allocation granularity already
|
||||
// exceeds file_size + SIMDJSON_PADDING, the OS page zero-fill provides
|
||||
// the padding and we use a simple MapViewOfFile3 call.
|
||||
// 2. Otherwise we reserve a contiguous placeholder region via VirtualAlloc2,
|
||||
// split it at the granularity-aligned file boundary, map the file into
|
||||
// the first part, and commit zero pages for the second part (padding).
|
||||
simdjson_inline padded_memory_map::padded_memory_map(const char *filename) noexcept {
|
||||
HANDLE file_handle = ::CreateFileA(
|
||||
filename, GENERIC_READ,
|
||||
FILE_SHARE_READ | FILE_SHARE_WRITE | FILE_SHARE_DELETE,
|
||||
NULL, OPEN_EXISTING, FILE_ATTRIBUTE_NORMAL, NULL);
|
||||
if (file_handle == INVALID_HANDLE_VALUE) {
|
||||
return;
|
||||
}
|
||||
LARGE_INTEGER file_size_li;
|
||||
if (!::GetFileSizeEx(file_handle, &file_size_li) || file_size_li.QuadPart < 0) {
|
||||
::CloseHandle(file_handle);
|
||||
return;
|
||||
}
|
||||
#if SIMDJSON_IS_32BITS
|
||||
if (static_cast<unsigned long long>(file_size_li.QuadPart) >
|
||||
static_cast<unsigned long long>(SIZE_MAX - simdjson::SIMDJSON_PADDING)) {
|
||||
::CloseHandle(file_handle);
|
||||
return;
|
||||
}
|
||||
#endif
|
||||
size = static_cast<size_t>(file_size_li.QuadPart);
|
||||
if (size == 0) {
|
||||
::CloseHandle(file_handle);
|
||||
return;
|
||||
}
|
||||
|
||||
HANDLE section = ::CreateFileMapping2(
|
||||
file_handle, NULL, FILE_MAP_READ, PAGE_READONLY,
|
||||
0, 0, NULL, NULL, 0);
|
||||
::CloseHandle(file_handle);
|
||||
if (section == NULL) {
|
||||
return;
|
||||
}
|
||||
|
||||
SYSTEM_INFO si;
|
||||
::GetSystemInfo(&si);
|
||||
const size_t granularity = static_cast<size_t>(si.dwAllocationGranularity);
|
||||
const size_t file_region = (size + granularity - 1) & ~(granularity - 1);
|
||||
const size_t total_needed = size + simdjson::SIMDJSON_PADDING;
|
||||
|
||||
if (file_region >= total_needed) {
|
||||
// The zero-fill in the last page already covers the padding.
|
||||
PVOID view = ::MapViewOfFile3(
|
||||
section, ::GetCurrentProcess(), NULL, 0, 0,
|
||||
0, PAGE_READONLY, NULL, 0);
|
||||
::CloseHandle(section);
|
||||
if (view != NULL) {
|
||||
data = static_cast<const char *>(view);
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
// We need extra zero pages beyond the file region. Use the placeholder API
|
||||
// to get a contiguous virtual address range spanning both the file mapping
|
||||
// and the zero-filled padding.
|
||||
const size_t padding_region =
|
||||
((total_needed - file_region) + granularity - 1) & ~(granularity - 1);
|
||||
const size_t reserve_size = file_region + padding_region;
|
||||
|
||||
// Reserve a contiguous placeholder.
|
||||
PVOID placeholder = ::VirtualAlloc2(
|
||||
::GetCurrentProcess(), NULL, reserve_size,
|
||||
MEM_RESERVE | MEM_RESERVE_PLACEHOLDER, PAGE_NOACCESS, NULL, 0);
|
||||
if (placeholder == NULL) {
|
||||
::CloseHandle(section);
|
||||
return;
|
||||
}
|
||||
|
||||
// Split into two placeholders at the file_region boundary.
|
||||
if (!::VirtualFree(placeholder, file_region,
|
||||
MEM_RELEASE | MEM_PRESERVE_PLACEHOLDER)) {
|
||||
::VirtualFree(placeholder, 0, MEM_RELEASE);
|
||||
::CloseHandle(section);
|
||||
return;
|
||||
}
|
||||
|
||||
// Map the file into the first placeholder.
|
||||
PVOID file_view = ::MapViewOfFile3(
|
||||
section, ::GetCurrentProcess(), placeholder, 0, file_region,
|
||||
MEM_REPLACE_PLACEHOLDER, PAGE_READONLY, NULL, 0);
|
||||
::CloseHandle(section);
|
||||
if (file_view == NULL) {
|
||||
::VirtualFree(placeholder, 0, MEM_RELEASE);
|
||||
::VirtualFree(static_cast<char *>(placeholder) + file_region,
|
||||
0, MEM_RELEASE);
|
||||
return;
|
||||
}
|
||||
|
||||
// Commit zero pages in the second placeholder (the padding).
|
||||
void *pad = static_cast<char *>(placeholder) + file_region;
|
||||
PVOID padding_ptr = ::VirtualAlloc2(
|
||||
::GetCurrentProcess(), pad, padding_region,
|
||||
MEM_REPLACE_PLACEHOLDER | MEM_COMMIT, PAGE_READONLY, NULL, 0);
|
||||
if (padding_ptr == NULL) {
|
||||
::UnmapViewOfFile(file_view);
|
||||
::VirtualFree(pad, 0, MEM_RELEASE);
|
||||
return;
|
||||
}
|
||||
|
||||
data = static_cast<const char *>(file_view);
|
||||
padding_view_ = padding_ptr;
|
||||
}
|
||||
|
||||
simdjson_inline padded_memory_map::~padded_memory_map() noexcept {
|
||||
if (data == nullptr) { return; }
|
||||
::UnmapViewOfFile(data);
|
||||
if (padding_view_ != nullptr) {
|
||||
::VirtualFree(padding_view_, 0, MEM_RELEASE);
|
||||
}
|
||||
}
|
||||
#endif // POSIX or _WIN32
|
||||
|
||||
simdjson_inline simdjson::padded_string_view padded_memory_map::view() const noexcept simdjson_lifetime_bound {
|
||||
if(!is_valid()) {
|
||||
@@ -565,8 +433,7 @@ simdjson_inline simdjson::padded_string_view padded_memory_map::view() const noe
|
||||
simdjson_inline bool padded_memory_map::is_valid() const noexcept {
|
||||
return data != nullptr;
|
||||
}
|
||||
|
||||
#endif // SIMDJSON_HAS_PADDED_MEMORY_MAP
|
||||
#endif // _WIN32
|
||||
|
||||
} // namespace simdjson
|
||||
|
||||
|
||||
@@ -277,26 +277,11 @@ inline std::ostream& operator<<(std::ostream& out, const padded_string& s) { ret
|
||||
inline std::ostream& operator<<(std::ostream& out, simdjson_result<padded_string> &s) noexcept(false) { return out << s.value(); }
|
||||
#endif
|
||||
|
||||
#if SIMDJSON_HAS_PADDED_MEMORY_MAP
|
||||
|
||||
#ifndef _WIN32
|
||||
/**
|
||||
* A class representing a memory-mapped file with padding.
|
||||
*
|
||||
* On POSIX systems (Linux, macOS, BSD, ...), this uses `mmap` to map the file
|
||||
* contents directly into memory, which is efficient for large files (no copy).
|
||||
*
|
||||
* On Windows, this class is disabled by default and must be opted into at
|
||||
* build time by defining `SIMDJSON_ENABLE_MEMORY_FILE_MAPPING_ON_WINDOWS=1`. When
|
||||
* enabled, `<windows.h>` must also be included before `<simdjson.h>` and
|
||||
* the compilation must target Windows 10, version 1803 or later. The
|
||||
* Windows implementation uses the modern memory APIs (`VirtualAlloc2`,
|
||||
* `CreateFileMapping2`, `MapViewOfFile3`) with the placeholder virtual
|
||||
* memory mechanism to always achieve true zero-copy mapping with
|
||||
* contiguous zero-filled padding.
|
||||
*
|
||||
* Either way, the resulting `padded_string_view` carries at least
|
||||
* `SIMDJSON_PADDING` bytes of accessible zero-filled padding after the file
|
||||
* content, so it can be consumed directly by the simdjson parsers (including
|
||||
* `parse_many` / `iterate_many`).
|
||||
* It is only available on non-Windows platforms, as Windows has different APIs for memory mapping.
|
||||
*/
|
||||
class padded_memory_map {
|
||||
public:
|
||||
@@ -304,11 +289,9 @@ public:
|
||||
* Create a new padded memory map for the given file.
|
||||
* After creating the memory map, you can call view() to get a padded_string_view of the file content.
|
||||
* The memory map will be automatically released when the padded_memory_map instance is destroyed.
|
||||
* On POSIX systems, the file content is not copied, so this is efficient for large files.
|
||||
* On Windows, the file is mapped into memory via `MapViewOfFile3` (zero-copy).
|
||||
* In all cases, the file must remain unchanged while the memory map is in use.
|
||||
* In case of error (e.g., file not found, permission denied, etc.), the memory map will be
|
||||
* invalid and view() will return an empty view.
|
||||
* Note that the file content is not copied, so this is efficient for large files. However,
|
||||
* the file must remain unchanged while the memory map is in use. In case of error (e.g., file not found,
|
||||
* permission denied, etc.), the memory map will be invalid and view() will return an empty view.
|
||||
* You can check if the memory map is valid by calling is_valid() before using view().
|
||||
*
|
||||
* @param filename the path to the file to memory-map.
|
||||
@@ -345,14 +328,8 @@ private:
|
||||
padded_memory_map &operator=(const padded_memory_map &) = delete;
|
||||
const char *data{nullptr};
|
||||
size_t size{0};
|
||||
#ifdef _WIN32
|
||||
// When the file ends near an allocation-granularity boundary, we use the
|
||||
// placeholder API to append zero-filled padding pages. This pointer tracks
|
||||
// that region so the destructor can release it with VirtualFree.
|
||||
void *padding_view_{nullptr};
|
||||
#endif
|
||||
};
|
||||
#endif // SIMDJSON_HAS_PADDED_MEMORY_MAP
|
||||
#endif // _WIN32
|
||||
|
||||
|
||||
|
||||
|
||||
@@ -7,7 +7,7 @@
|
||||
#include <cstring> /* memcmp */
|
||||
|
||||
// for page size computation.
|
||||
#if SIMDJSON_HAS_UNISTD_H
|
||||
#if defined(__unix__) || defined(__APPLE__) || defined(__linux__)
|
||||
#include <unistd.h>
|
||||
#if defined(__APPLE__)
|
||||
#include <sys/sysctl.h>
|
||||
@@ -113,7 +113,7 @@ inline uint32_t get_page_size() noexcept {
|
||||
return static_cast<std::uint32_t>(si.dwPageSize);
|
||||
}();
|
||||
return cached;
|
||||
#elif SIMDJSON_HAS_UNISTD_H
|
||||
#elif defined(__unix__) || defined(__APPLE__) || defined(__linux__)
|
||||
static const uint32_t cached = []() -> uint32_t {
|
||||
long page_size = sysconf(_SC_PAGESIZE);
|
||||
if (page_size > 0) {
|
||||
|
||||
@@ -285,53 +285,5 @@ using std::size_t;
|
||||
#endif
|
||||
#endif
|
||||
|
||||
#ifndef SIMDJSON_HAS_UNISTD_H
|
||||
#if defined(__unix__) || defined(__APPLE__) || defined(__linux__)
|
||||
#define SIMDJSON_HAS_UNISTD_H 1
|
||||
#else
|
||||
#define SIMDJSON_HAS_UNISTD_H 0
|
||||
#endif
|
||||
#endif
|
||||
|
||||
// padded_memory_map availability.
|
||||
//
|
||||
// On POSIX platforms the class is always available: the implementation uses
|
||||
// `mmap` (and a trailing anonymous page for padding) from <sys/mman.h>.
|
||||
//
|
||||
// On Windows the class is disabled by default and must be explicitly
|
||||
// opted into by defining `SIMDJSON_ENABLE_MEMORY_FILE_MAPPING_ON_WINDOWS=1`. Enabling
|
||||
// it requires:
|
||||
// 1. `<windows.h>` has been included *before* `<simdjson.h>` (so that
|
||||
// this header can see the Win32 types and the `_WINDOWS_` include
|
||||
// guard),
|
||||
// 2. the compilation targets Windows 10, version 1803 or later
|
||||
// (i.e. `NTDDI_VERSION >= NTDDI_WIN10_RS4`, `0x0A000005`). This is
|
||||
// required because the implementation relies on the modern memory
|
||||
// APIs introduced with that version (`CreateFileMapping2` /
|
||||
// `MapViewOfFile3`),
|
||||
// 3. the link step pulls in an import library that exports those APIs,
|
||||
// typically `onecore.lib` (or `mincore.lib`).
|
||||
//
|
||||
// The `SIMDJSON_ENABLE_MEMORY_FILE_MAPPING_ON_WINDOWS` CMake option arranges (1)-(3)
|
||||
// automatically when building simdjson with its own CMake. Consumers using
|
||||
// simdjson as a pre-built library are responsible for setting the macro,
|
||||
// the Windows version macros, and the link library themselves.
|
||||
//
|
||||
// If the opt-in conditions are not met on Windows, `padded_memory_map`
|
||||
// simply does not exist — any attempt to use it fails at compile time
|
||||
// with an "unknown identifier" diagnostic rather than silently degrading.
|
||||
//
|
||||
// The SIMDJSON_HAS_PADDED_MEMORY_MAP macro reflects whether the class is
|
||||
// available in the current translation unit. Users may test this macro to
|
||||
// conditionally compile code that depends on padded_memory_map.
|
||||
#ifndef SIMDJSON_HAS_PADDED_MEMORY_MAP
|
||||
#if defined(__unix__) || defined(__APPLE__) || defined(__linux__)
|
||||
#define SIMDJSON_HAS_PADDED_MEMORY_MAP 1
|
||||
#elif defined(_WINDOWS_) && defined(SIMDJSON_ENABLE_MEMORY_FILE_MAPPING_ON_WINDOWS) && SIMDJSON_ENABLE_MEMORY_FILE_MAPPING_ON_WINDOWS
|
||||
#define SIMDJSON_HAS_PADDED_MEMORY_MAP 1
|
||||
#else
|
||||
#define SIMDJSON_HAS_PADDED_MEMORY_MAP 0
|
||||
#endif
|
||||
#endif
|
||||
|
||||
#endif // SIMDJSON_PORTABILITY_H
|
||||
|
||||
+7
-203
@@ -4356,55 +4356,10 @@ inline std::ostream& operator<<(std::ostream& out, simdjson_result<padded_string
|
||||
#endif
|
||||
|
||||
|
||||
// padded_memory_map availability.
|
||||
//
|
||||
// On POSIX platforms the class is always available: the implementation uses
|
||||
// `mmap` (and a trailing anonymous page for padding) from <sys/mman.h>.
|
||||
//
|
||||
// On Windows the class is only available when all of the following hold:
|
||||
// 1. <windows.h> has been included *before* <simdjson.h> (so that this
|
||||
// header can see the Win32 types and the `_WINDOWS_` include guard),
|
||||
// 2. the compilation targets Windows 11 or later (NTDDI_VERSION
|
||||
// >= NTDDI_WIN10_CO, 0x0A00000B). This is required because the
|
||||
// implementation relies on the modern memory APIs introduced with
|
||||
// that version (CreateFileMapping2 / MapViewOfFile3).
|
||||
//
|
||||
// If those conditions are not met on Windows, `padded_memory_map` simply
|
||||
// does not exist — any attempt to use it fails at compile time with an
|
||||
// "unknown identifier" diagnostic rather than silently degrading.
|
||||
//
|
||||
// The SIMDJSON_HAS_PADDED_MEMORY_MAP macro reflects whether the class is
|
||||
// available in the current translation unit. Users may test this macro to
|
||||
// conditionally compile code that depends on padded_memory_map.
|
||||
#ifndef SIMDJSON_HAS_PADDED_MEMORY_MAP
|
||||
#if !defined(_WIN32)
|
||||
#define SIMDJSON_HAS_PADDED_MEMORY_MAP 1
|
||||
#elif defined(_WINDOWS_) && defined(NTDDI_VERSION) && (NTDDI_VERSION >= 0x0A00000B /* NTDDI_WIN10_CO — Windows 11 */)
|
||||
#define SIMDJSON_HAS_PADDED_MEMORY_MAP 1
|
||||
#else
|
||||
#define SIMDJSON_HAS_PADDED_MEMORY_MAP 0
|
||||
#endif
|
||||
#endif
|
||||
|
||||
#if SIMDJSON_HAS_PADDED_MEMORY_MAP
|
||||
#ifndef _WIN32
|
||||
/**
|
||||
* A class representing a memory-mapped file with padding.
|
||||
*
|
||||
* On POSIX systems (Linux, macOS, BSD, ...), this uses `mmap` to map the file
|
||||
* contents directly into memory, which is efficient for large files (no copy).
|
||||
*
|
||||
* On Windows, this class is only available when `<windows.h>` is included
|
||||
* before `<simdjson.h>` and the compilation targets Windows 11 or later
|
||||
* (NTDDI_VERSION >= NTDDI_WIN10_CO). The Windows implementation uses the
|
||||
* modern memory APIs (`CreateFileMapping2` / `MapViewOfFile3`) to map the
|
||||
* file with true zero-copy semantics whenever the last page of the file
|
||||
* provides enough trailing zero-fill for SIMDJSON_PADDING bytes; otherwise
|
||||
* it falls back to a heap-allocated padded buffer populated with `ReadFile`.
|
||||
*
|
||||
* Either way, the resulting `padded_string_view` carries at least
|
||||
* `SIMDJSON_PADDING` bytes of accessible zero-filled padding after the file
|
||||
* content, so it can be consumed directly by the simdjson parsers (including
|
||||
* `parse_many` / `iterate_many`).
|
||||
* It is only available on non-Windows platforms, as Windows has different APIs for memory mapping.
|
||||
*/
|
||||
class padded_memory_map {
|
||||
public:
|
||||
@@ -4412,12 +4367,9 @@ public:
|
||||
* Create a new padded memory map for the given file.
|
||||
* After creating the memory map, you can call view() to get a padded_string_view of the file content.
|
||||
* The memory map will be automatically released when the padded_memory_map instance is destroyed.
|
||||
* On POSIX systems, the file content is not copied, so this is efficient for large files.
|
||||
* On Windows, the file is mapped into memory via `MapViewOfFile3` whenever possible
|
||||
* (zero-copy) and otherwise read into a heap-allocated padded buffer.
|
||||
* In all cases, the file must remain unchanged while the memory map is in use.
|
||||
* In case of error (e.g., file not found, permission denied, etc.), the memory map will be
|
||||
* invalid and view() will return an empty view.
|
||||
* Note that the file content is not copied, so this is efficient for large files. However,
|
||||
* the file must remain unchanged while the memory map is in use. In case of error (e.g., file not found,
|
||||
* permission denied, etc.), the memory map will be invalid and view() will return an empty view.
|
||||
* You can check if the memory map is valid by calling is_valid() before using view().
|
||||
*
|
||||
* @param filename the path to the file to memory-map.
|
||||
@@ -4454,14 +4406,8 @@ private:
|
||||
padded_memory_map &operator=(const padded_memory_map &) = delete;
|
||||
const char *data{nullptr};
|
||||
size_t size{0};
|
||||
#ifdef _WIN32
|
||||
// On Windows the underlying storage may either be a memory-mapped view
|
||||
// (released with UnmapViewOfFile) or a heap-allocated padded buffer
|
||||
// (released with delete[]). This flag distinguishes the two cases.
|
||||
bool owns_heap_buffer_{false};
|
||||
#endif
|
||||
};
|
||||
#endif // SIMDJSON_HAS_PADDED_MEMORY_MAP
|
||||
#endif // _WIN32
|
||||
|
||||
|
||||
|
||||
@@ -4754,11 +4700,6 @@ inline padded_string_view pad_with_reserve(std::string& s) noexcept {
|
||||
#include <sys/stat.h>
|
||||
#include <unistd.h>
|
||||
#endif
|
||||
// On Windows, `padded_memory_map` (when it is enabled) depends on types and
|
||||
// functions declared in <windows.h>. We deliberately do NOT include that
|
||||
// header here: users of simdjson who want `padded_memory_map` on Windows
|
||||
// must include <windows.h> themselves *before* including this header. See
|
||||
// padded_string.h for the detection logic.
|
||||
|
||||
namespace simdjson {
|
||||
namespace internal {
|
||||
@@ -5127,8 +5068,6 @@ inline bool padded_string_builder::reserve(size_t additional) noexcept {
|
||||
}
|
||||
|
||||
|
||||
#if SIMDJSON_HAS_PADDED_MEMORY_MAP
|
||||
|
||||
#ifndef _WIN32
|
||||
simdjson_inline padded_memory_map::padded_memory_map(const char *filename) noexcept {
|
||||
|
||||
@@ -5165,141 +5104,7 @@ simdjson_inline padded_memory_map::~padded_memory_map() noexcept {
|
||||
munmap(const_cast<char *>(data), size + simdjson::SIMDJSON_PADDING);
|
||||
}
|
||||
}
|
||||
#else // _WIN32
|
||||
// Windows 11+ implementation.
|
||||
//
|
||||
// We use the modern Windows memory APIs (CreateFileMapping2 + MapViewOfFile3,
|
||||
// available since Windows 10 1803 and gated on Windows 11 in our build) to
|
||||
// map the file directly into the process address space with zero copies.
|
||||
//
|
||||
// Windows guarantees that after a file view is mapped, any bytes in the
|
||||
// trailing partial page beyond the end of the file are zero-filled. As long
|
||||
// as the file does not end exactly on (or within SIMDJSON_PADDING bytes of)
|
||||
// a page boundary, we therefore get SIMDJSON_PADDING accessible zero bytes
|
||||
// for free at the tail of the view. In the rare edge cases where the tail
|
||||
// is not large enough (about 1.5% of file sizes if sizes were uniformly
|
||||
// distributed), we fall back to reading the file into a heap-allocated
|
||||
// padded buffer. That fallback is still correct — it just performs one
|
||||
// memory copy instead of a zero-copy mapping.
|
||||
simdjson_inline padded_memory_map::padded_memory_map(const char *filename) noexcept {
|
||||
HANDLE file_handle = ::CreateFileA(
|
||||
filename, GENERIC_READ,
|
||||
FILE_SHARE_READ | FILE_SHARE_WRITE | FILE_SHARE_DELETE,
|
||||
NULL, OPEN_EXISTING, FILE_ATTRIBUTE_NORMAL, NULL);
|
||||
if (file_handle == INVALID_HANDLE_VALUE) {
|
||||
return; // file not found or cannot be opened
|
||||
}
|
||||
LARGE_INTEGER file_size_li;
|
||||
if (!::GetFileSizeEx(file_handle, &file_size_li) || file_size_li.QuadPart < 0) {
|
||||
::CloseHandle(file_handle);
|
||||
return; // failed to get file size
|
||||
}
|
||||
#if SIMDJSON_IS_32BITS
|
||||
if (static_cast<unsigned long long>(file_size_li.QuadPart) >
|
||||
static_cast<unsigned long long>(SIZE_MAX - simdjson::SIMDJSON_PADDING)) {
|
||||
::CloseHandle(file_handle);
|
||||
return; // file too large to map on a 32-bit system
|
||||
}
|
||||
#endif
|
||||
size = static_cast<size_t>(file_size_li.QuadPart);
|
||||
|
||||
// Fast zero-copy path: only usable when the last partial page of the file
|
||||
// gives us at least SIMDJSON_PADDING bytes of zero-filled slack.
|
||||
if (size > 0) {
|
||||
SYSTEM_INFO sys_info;
|
||||
::GetSystemInfo(&sys_info);
|
||||
const size_t page_size = static_cast<size_t>(sys_info.dwPageSize);
|
||||
const size_t tail_in_page = size % page_size;
|
||||
const size_t tail_zero_fill = (tail_in_page == 0)
|
||||
? size_t{0}
|
||||
: (page_size - tail_in_page);
|
||||
|
||||
if (tail_zero_fill >= simdjson::SIMDJSON_PADDING) {
|
||||
// Create the section with the new CreateFileMapping2 API.
|
||||
HANDLE mapping = ::CreateFileMapping2(
|
||||
file_handle, /*SecurityAttributes=*/ NULL,
|
||||
/*DesiredAccess=*/ FILE_MAP_READ,
|
||||
/*PageProtection=*/ PAGE_READONLY,
|
||||
/*AllocationAttributes=*/ 0,
|
||||
/*MaximumSize=*/ 0, // 0 => entire file
|
||||
/*Name=*/ NULL,
|
||||
/*ExtendedParameters=*/ NULL, /*ParameterCount=*/ 0);
|
||||
if (mapping != NULL) {
|
||||
// Map the view with the new MapViewOfFile3 API.
|
||||
PVOID view_ptr = ::MapViewOfFile3(
|
||||
mapping, ::GetCurrentProcess(),
|
||||
/*BaseAddress=*/ NULL,
|
||||
/*Offset=*/ 0,
|
||||
/*ViewSize=*/ size,
|
||||
/*AllocationType=*/ 0,
|
||||
/*PageProtection=*/ PAGE_READONLY,
|
||||
/*ExtendedParameters=*/ NULL, /*ParameterCount=*/ 0);
|
||||
::CloseHandle(mapping);
|
||||
if (view_ptr != NULL) {
|
||||
::CloseHandle(file_handle);
|
||||
data = static_cast<const char *>(view_ptr);
|
||||
owns_heap_buffer_ = false;
|
||||
return;
|
||||
}
|
||||
}
|
||||
// Fall through to the buffered-read fallback if the mapping failed.
|
||||
}
|
||||
}
|
||||
|
||||
// Fallback path: the file ends too close to a page boundary (or the
|
||||
// mapping APIs refused) — read the file contents into a heap-allocated
|
||||
// padded buffer. This preserves the class' padding invariant at the cost
|
||||
// of one copy.
|
||||
size_t total_size = size + simdjson::SIMDJSON_PADDING;
|
||||
if (total_size < size) { // overflow guard
|
||||
::CloseHandle(file_handle);
|
||||
size = 0;
|
||||
return;
|
||||
}
|
||||
char *buffer = new (std::nothrow) char[total_size];
|
||||
if (buffer == nullptr) {
|
||||
::CloseHandle(file_handle);
|
||||
size = 0;
|
||||
return;
|
||||
}
|
||||
size_t total_read = 0;
|
||||
while (total_read < size) {
|
||||
size_t remaining = size - total_read;
|
||||
const size_t chunk_limit = static_cast<size_t>(0x40000000UL); // 1 GiB per call
|
||||
DWORD to_read = remaining > chunk_limit
|
||||
? static_cast<DWORD>(chunk_limit)
|
||||
: static_cast<DWORD>(remaining);
|
||||
DWORD bytes_read = 0;
|
||||
if (!::ReadFile(file_handle, buffer + total_read, to_read, &bytes_read, NULL)) {
|
||||
delete[] buffer;
|
||||
::CloseHandle(file_handle);
|
||||
size = 0;
|
||||
return;
|
||||
}
|
||||
if (bytes_read == 0) {
|
||||
// Unexpected EOF: the file shrank while we were reading it.
|
||||
delete[] buffer;
|
||||
::CloseHandle(file_handle);
|
||||
size = 0;
|
||||
return;
|
||||
}
|
||||
total_read += bytes_read;
|
||||
}
|
||||
std::memset(buffer + size, 0, simdjson::SIMDJSON_PADDING);
|
||||
data = buffer;
|
||||
owns_heap_buffer_ = true;
|
||||
::CloseHandle(file_handle);
|
||||
}
|
||||
|
||||
simdjson_inline padded_memory_map::~padded_memory_map() noexcept {
|
||||
if (data == nullptr) { return; }
|
||||
if (owns_heap_buffer_) {
|
||||
delete[] const_cast<char *>(data);
|
||||
} else {
|
||||
::UnmapViewOfFile(data);
|
||||
}
|
||||
}
|
||||
#endif // _WIN32
|
||||
|
||||
simdjson_inline simdjson::padded_string_view padded_memory_map::view() const noexcept simdjson_lifetime_bound {
|
||||
if(!is_valid()) {
|
||||
@@ -5311,8 +5116,7 @@ simdjson_inline simdjson::padded_string_view padded_memory_map::view() const noe
|
||||
simdjson_inline bool padded_memory_map::is_valid() const noexcept {
|
||||
return data != nullptr;
|
||||
}
|
||||
|
||||
#endif // SIMDJSON_HAS_PADDED_MEMORY_MAP
|
||||
#endif // _WIN32
|
||||
|
||||
} // namespace simdjson
|
||||
|
||||
|
||||
@@ -52,13 +52,17 @@ POSSIBILITY OF SUCH DAMAGE.
|
||||
#include <cstdlib>
|
||||
#if defined(_MSC_VER)
|
||||
#include <intrin.h>
|
||||
#elif defined(HAVE_GCC_GET_CPUID) && defined(USE_GCC_GET_CPUID)
|
||||
#elif (defined(HAVE_GCC_GET_CPUID) && defined(USE_GCC_GET_CPUID)) || defined(__FILC__)
|
||||
#include <cpuid.h>
|
||||
#endif
|
||||
#if defined(__loongarch__) && defined(__linux__)
|
||||
#include <sys/auxv.h>
|
||||
#endif
|
||||
|
||||
#ifdef __FILC__
|
||||
#include <stdfil.h>
|
||||
#endif
|
||||
|
||||
namespace simdjson {
|
||||
namespace internal {
|
||||
|
||||
@@ -109,7 +113,7 @@ static inline void cpuid(uint32_t *eax, uint32_t *ebx, uint32_t *ecx,
|
||||
*ebx = cpu_info[1];
|
||||
*ecx = cpu_info[2];
|
||||
*edx = cpu_info[3];
|
||||
#elif defined(HAVE_GCC_GET_CPUID) && defined(USE_GCC_GET_CPUID)
|
||||
#elif (defined(HAVE_GCC_GET_CPUID) && defined(USE_GCC_GET_CPUID)) || defined(__FILC__)
|
||||
uint32_t level = *eax;
|
||||
__get_cpuid(level, eax, ebx, ecx, edx);
|
||||
#else
|
||||
@@ -126,6 +130,8 @@ static inline void cpuid(uint32_t *eax, uint32_t *ebx, uint32_t *ecx,
|
||||
static inline uint64_t xgetbv() {
|
||||
#if defined(_MSC_VER)
|
||||
return _xgetbv(0);
|
||||
#elif defined(__FILC__)
|
||||
return zxgetbv();
|
||||
#else
|
||||
uint32_t xcr0_lo, xcr0_hi;
|
||||
asm volatile("xgetbv\n\t" : "=a" (xcr0_lo), "=d" (xcr0_hi) : "c" (0));
|
||||
|
||||
@@ -127,14 +127,16 @@ endif()
|
||||
# 1. Visual Studio 2022 v17.6 or later
|
||||
# 2. GCC v14.0.0 or later (GCC v13.0.0 cannot handle pipe operator of lambda)
|
||||
# 3. Clang v15.0.0 or later (certain version C++ headers occur error when compiling)
|
||||
# 4. or if we are targeting C++20 or better
|
||||
if(
|
||||
(MSVC AND MSVC_VERSION LESS 1930) OR
|
||||
(MSVC AND MSVC_VERSION GREATER_EQUAL 1930) OR
|
||||
(CMAKE_CXX_COMPILER_ID STREQUAL "GNU" AND CMAKE_CXX_COMPILER_VERSION VERSION_GREATER_EQUAL "14.0.0") OR
|
||||
(CMAKE_CXX_COMPILER_ID MATCHES "Clang" AND CMAKE_CXX_COMPILER_VERSION VERSION_GREATER_EQUAL "15.0.0")
|
||||
(CMAKE_CXX_COMPILER_ID MATCHES "Clang" AND CMAKE_CXX_COMPILER_VERSION VERSION_GREATER_EQUAL "15.0.0") OR
|
||||
(CMAKE_CXX_STANDARD GREATER_EQUAL 20)
|
||||
)
|
||||
message(STATUS "compiler id: ${CMAKE_CXX_COMPILER_ID} version: ${CMAKE_CXX_COMPILER_VERSION}")
|
||||
add_cpp_test(ranges_test LABELS dom acceptance per_implementation)
|
||||
if(NOT SIMDJSON_STATIC_REFLECTION)
|
||||
if(NOT SIMDJSON_STATIC_REFLECTION AND NOT CMAKE_CXX_STANDARD GREATER_EQUAL 20)
|
||||
set_target_properties(ranges_test PROPERTIES CXX_STANDARD 20 CXX_STANDARD_REQUIRED ON CXX_EXTENSIONS OFF)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
+8
-102
@@ -1,29 +1,13 @@
|
||||
// On Windows, padded_memory_map is an opt-in feature gated on the
|
||||
// SIMDJSON_ENABLE_MEMORY_FILE_MAPPING_ON_WINDOWS macro. When that macro is set, the
|
||||
// consumer must also include <windows.h> before <simdjson.h>. We include
|
||||
// the Win32 header here so that -- in configurations that turned the
|
||||
// feature on -- the test actually exercises the Windows path.
|
||||
#ifdef _WIN32
|
||||
#ifndef WIN32_LEAN_AND_MEAN
|
||||
#define WIN32_LEAN_AND_MEAN
|
||||
#endif
|
||||
#ifndef NOMINMAX
|
||||
#define NOMINMAX
|
||||
#endif
|
||||
#include <windows.h>
|
||||
#endif
|
||||
#include <cstdlib>
|
||||
// This test is not supported on Windows because it relies on POSIX APIs like
|
||||
// mmap. Please run it on a POSIX-compliant system.
|
||||
int main() { return EXIT_SUCCESS; }
|
||||
#else
|
||||
|
||||
#include "simdjson.h"
|
||||
#include "test_macros.h"
|
||||
|
||||
// When SIMDJSON_HAS_PADDED_MEMORY_MAP is 0 (e.g. Windows builds without
|
||||
// SIMDJSON_ENABLE_MEMORY_FILE_MAPPING_ON_WINDOWS, or MinGW configurations that lack
|
||||
// the required SDK gating), compile the test body out and make main()
|
||||
// report success so the test suite still runs as a no-op. This is not a
|
||||
// silent downgrade: users who want the Windows path must explicitly
|
||||
// enable the CMake option `SIMDJSON_ENABLE_MEMORY_FILE_MAPPING_ON_WINDOWS`.
|
||||
#if SIMDJSON_HAS_PADDED_MEMORY_MAP
|
||||
|
||||
#if SIMDJSON_EXCEPTIONS
|
||||
bool test_memory_map_exception() {
|
||||
TEST_START();
|
||||
@@ -53,89 +37,11 @@ bool test_memory_map_noexception() {
|
||||
TEST_SUCCEED();
|
||||
}
|
||||
|
||||
// Verifies that padded_memory_map can feed a streaming parser (iterate_many)
|
||||
// with JSON documents read from a file. This exercises the API that parse_many
|
||||
// / iterate_many users typically want: no extra copy on POSIX, portable fallback
|
||||
// on Windows. The AMAZON_CELLPHONES_NDJSON resource is an NDJSON file so it is
|
||||
// a realistic stress-test for streaming from a memory-mapped file.
|
||||
bool test_memory_map_iterate_many() {
|
||||
TEST_START();
|
||||
simdjson::padded_memory_map map(AMAZON_CELLPHONES_NDJSON);
|
||||
if (!map.is_valid()) {
|
||||
std::cerr << "Failed to memory-map the file " << AMAZON_CELLPHONES_NDJSON << std::endl;
|
||||
return false;
|
||||
}
|
||||
simdjson::padded_string_view view = map.view();
|
||||
simdjson::ondemand::parser parser;
|
||||
simdjson::ondemand::document_stream stream;
|
||||
ASSERT_SUCCESS( parser.iterate_many(view).get(stream) );
|
||||
size_t count = 0;
|
||||
for (auto doc : stream) {
|
||||
ASSERT_SUCCESS( doc.error() );
|
||||
count++;
|
||||
}
|
||||
if (count == 0) {
|
||||
std::cerr << "Expected at least one document in " << AMAZON_CELLPHONES_NDJSON << std::endl;
|
||||
return false;
|
||||
}
|
||||
TEST_SUCCEED();
|
||||
}
|
||||
|
||||
// Verifies that padded_memory_map also works with the DOM streaming parser
|
||||
// (parse_many). Same rationale as the ondemand variant above.
|
||||
bool test_memory_map_parse_many() {
|
||||
TEST_START();
|
||||
simdjson::padded_memory_map map(AMAZON_CELLPHONES_NDJSON);
|
||||
if (!map.is_valid()) {
|
||||
std::cerr << "Failed to memory-map the file " << AMAZON_CELLPHONES_NDJSON << std::endl;
|
||||
return false;
|
||||
}
|
||||
simdjson::padded_string_view view = map.view();
|
||||
simdjson::dom::parser parser;
|
||||
simdjson::dom::document_stream stream;
|
||||
ASSERT_SUCCESS( parser.parse_many(view).get(stream) );
|
||||
size_t count = 0;
|
||||
for (auto doc : stream) {
|
||||
ASSERT_SUCCESS( doc.error() );
|
||||
count++;
|
||||
}
|
||||
if (count == 0) {
|
||||
std::cerr << "Expected at least one document in " << AMAZON_CELLPHONES_NDJSON << std::endl;
|
||||
return false;
|
||||
}
|
||||
TEST_SUCCEED();
|
||||
}
|
||||
|
||||
// Ensures that trying to memory-map a file that does not exist leaves the map
|
||||
// in the "invalid" state rather than crashing. This is important on Windows
|
||||
// where the underlying implementation path differs from POSIX.
|
||||
bool test_memory_map_missing_file() {
|
||||
TEST_START();
|
||||
simdjson::padded_memory_map map("this_file_definitely_does_not_exist_123456789.json");
|
||||
if (map.is_valid()) {
|
||||
std::cerr << "Expected is_valid() == false for missing file" << std::endl;
|
||||
return false;
|
||||
}
|
||||
TEST_SUCCEED();
|
||||
}
|
||||
|
||||
#endif // SIMDJSON_HAS_PADDED_MEMORY_MAP
|
||||
|
||||
int main() {
|
||||
#if SIMDJSON_HAS_PADDED_MEMORY_MAP
|
||||
bool ok = true;
|
||||
#if SIMDJSON_EXCEPTIONS
|
||||
ok = ok && test_memory_map_exception();
|
||||
#endif
|
||||
ok = ok && test_memory_map_noexception();
|
||||
ok = ok && test_memory_map_iterate_many();
|
||||
ok = ok && test_memory_map_parse_many();
|
||||
ok = ok && test_memory_map_missing_file();
|
||||
return ok ? EXIT_SUCCESS : EXIT_FAILURE;
|
||||
return (test_memory_map_exception() && test_memory_map_noexception()) ? EXIT_SUCCESS : EXIT_FAILURE;
|
||||
#else
|
||||
std::cout << "padded_memory_map is disabled in this configuration; "
|
||||
"set SIMDJSON_ENABLE_MEMORY_FILE_MAPPING_ON_WINDOWS=ON in CMake to "
|
||||
"enable it on Windows. Test skipped." << std::endl;
|
||||
return EXIT_SUCCESS;
|
||||
return test_memory_map_noexception() ? EXIT_SUCCESS : EXIT_FAILURE;
|
||||
#endif
|
||||
}
|
||||
#endif
|
||||
@@ -25,7 +25,6 @@ add_cpp_test(ondemand_misc_tests LABELS ondemand acceptance
|
||||
add_cpp_test(ondemand_number_tests LABELS ondemand acceptance per_implementation)
|
||||
add_cpp_test(ondemand_number_in_string_tests LABELS ondemand acceptance per_implementation)
|
||||
add_cpp_test(ondemand_object_tests LABELS ondemand acceptance per_implementation)
|
||||
add_cpp_test(ondemand_object_find_field_tests LABELS ondemand acceptance per_implementation)
|
||||
add_cpp_test(ondemand_object_error_tests LABELS ondemand acceptance per_implementation)
|
||||
add_cpp_test(ondemand_ordering_tests LABELS ondemand acceptance per_implementation)
|
||||
add_cpp_test(ondemand_parse_api_tests LABELS ondemand acceptance per_implementation)
|
||||
@@ -46,6 +45,25 @@ if(NOT SIMDJSON_SANITIZE)
|
||||
add_cpp_test(ondemand_cacheline LABELS ondemand acceptance per_implementation)
|
||||
endif()
|
||||
|
||||
# Add the tests if we're on:
|
||||
# 1. Visual Studio 2022 v17.6 or later
|
||||
# 2. GCC v14.0.0 or later (GCC v13.0.0 cannot handle pipe operator of lambda)
|
||||
# 3. Clang v15.0.0 or later (certain version C++ headers occur error when compiling)
|
||||
# 4. or if we are targeting C++20 or better
|
||||
if(
|
||||
(MSVC AND MSVC_VERSION GREATER_EQUAL 1930) OR
|
||||
(CMAKE_CXX_COMPILER_ID STREQUAL "GNU" AND CMAKE_CXX_COMPILER_VERSION VERSION_GREATER_EQUAL "14.0.0") OR
|
||||
(CMAKE_CXX_COMPILER_ID MATCHES "Clang" AND CMAKE_CXX_COMPILER_VERSION VERSION_GREATER_EQUAL "15.0.0") OR
|
||||
(CMAKE_CXX_STANDARD GREATER_EQUAL 20)
|
||||
)
|
||||
message(STATUS "compiler id: ${CMAKE_CXX_COMPILER_ID} version: ${CMAKE_CXX_COMPILER_VERSION}")
|
||||
add_cpp_test(ondemand_ranges_tests LABELS ondemand acceptance per_implementation)
|
||||
if(NOT SIMDJSON_STATIC_REFLECTION AND NOT CMAKE_CXX_STANDARD GREATER_EQUAL 20)
|
||||
set_target_properties(ondemand_ranges_tests PROPERTIES CXX_STANDARD 20 CXX_STANDARD_REQUIRED ON CXX_EXTENSIONS OFF)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
|
||||
if(HAVE_POSIX_FORK AND HAVE_POSIX_WAIT) # assert tests use fork and wait, which aren't on MSVC
|
||||
add_cpp_test(ondemand_assert_out_of_order_values LABELS assert per_implementation explicitonly ondemand)
|
||||
endif()
|
||||
|
||||
@@ -1,3 +1,12 @@
|
||||
#ifdef __FILC__
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
int main() {
|
||||
printf("This test is not relevant for FILC.\n");
|
||||
return EXIT_SUCCESS;
|
||||
}
|
||||
#else // This test is not relevant for FILC
|
||||
|
||||
#ifdef _WIN32
|
||||
#include <windows.h>
|
||||
#include <sysinfoapi.h>
|
||||
@@ -86,4 +95,6 @@ int main() {
|
||||
return EXIT_FAILURE;
|
||||
}
|
||||
return EXIT_SUCCESS;
|
||||
}
|
||||
}
|
||||
|
||||
#endif // This test is not relevant for FILC
|
||||
@@ -173,36 +173,6 @@ namespace object_tests {
|
||||
TEST_SUCCEED();
|
||||
}
|
||||
|
||||
#if SIMDJSON_SUPPORTS_CONCEPTS
|
||||
bool object_find_field_key_selector() {
|
||||
TEST_START();
|
||||
auto json = R"({ "name": "John", "age": 30, "city": "New York" })"_padded;
|
||||
constexpr std::array<std::string_view, 3> keys = {"name", "age", "city"};
|
||||
constexpr auto selector = ondemand::key_selector<3>(keys);
|
||||
|
||||
SUBTEST("ondemand::object with key_selector", test_ondemand_doc(json, [&](auto doc_result) {
|
||||
ondemand::object object;
|
||||
ASSERT_SUCCESS( doc_result.get(object) );
|
||||
|
||||
auto [index, value_result] = object.find_field(selector);
|
||||
ASSERT_TRUE(index < 3);
|
||||
ASSERT_SUCCESS(value_result);
|
||||
std::string_view str_val;
|
||||
ASSERT_SUCCESS(value_result.get(str_val));
|
||||
ASSERT_EQUAL(str_val, "John");
|
||||
|
||||
// Test that we can find different keys
|
||||
ASSERT_EQUAL(selector.index_of("name"), 0);
|
||||
ASSERT_EQUAL(selector.index_of("age"), 1);
|
||||
ASSERT_EQUAL(selector.index_of("city"), 2);
|
||||
ASSERT_EQUAL(selector.index_of("invalid"), 3); // Not found
|
||||
|
||||
return true;
|
||||
}));
|
||||
TEST_SUCCEED();
|
||||
}
|
||||
#endif
|
||||
|
||||
bool run() {
|
||||
return
|
||||
object_find_field_unordered() &&
|
||||
@@ -211,9 +181,6 @@ namespace object_tests {
|
||||
object_find_field() &&
|
||||
document_object_find_field() &&
|
||||
value_object_find_field() &&
|
||||
#if SIMDJSON_SUPPORTS_CONCEPTS
|
||||
object_find_field_key_selector() &&
|
||||
#endif
|
||||
true;
|
||||
}
|
||||
|
||||
|
||||
@@ -1,3 +1,12 @@
|
||||
#ifdef __FILC__
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
int main() {
|
||||
printf("This test is not relevant for FILC.\n");
|
||||
return EXIT_SUCCESS;
|
||||
}
|
||||
#else // This test is not relevant for FILC
|
||||
|
||||
#include "simdjson.h"
|
||||
#include "simdjson/padded_string_view.h"
|
||||
#include <cstdio>
|
||||
@@ -195,3 +204,5 @@ int main() {
|
||||
}
|
||||
|
||||
#endif // SIMDJSON_CPLUSPLUS17
|
||||
|
||||
#endif // This test is not relevant for FILC
|
||||
@@ -0,0 +1,319 @@
|
||||
#include <iostream>
|
||||
#include "simdjson.h"
|
||||
#include "test_macros.h"
|
||||
#include "test_main.h"
|
||||
|
||||
#if SIMDJSON_SUPPORTS_RANGES
|
||||
|
||||
#include <algorithm>
|
||||
#include <ranges>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
using namespace simdjson;
|
||||
|
||||
namespace ondemand_ranges_tests {
|
||||
|
||||
#if SIMDJSON_EXCEPTIONS
|
||||
|
||||
bool array_get_range_basic() {
|
||||
TEST_START();
|
||||
auto json = R"([10, 20, 30])"_padded;
|
||||
ondemand::parser parser;
|
||||
auto doc = parser.iterate(json);
|
||||
auto arr = doc.get_array();
|
||||
auto range = ondemand::get_range(arr);
|
||||
|
||||
std::vector<int64_t> values;
|
||||
for (auto elem : range) {
|
||||
values.push_back(int64_t(elem));
|
||||
}
|
||||
ASSERT_EQUAL(values.size(), size_t(3));
|
||||
ASSERT_EQUAL(values[0], int64_t(10));
|
||||
ASSERT_EQUAL(values[1], int64_t(20));
|
||||
ASSERT_EQUAL(values[2], int64_t(30));
|
||||
TEST_SUCCEED();
|
||||
}
|
||||
|
||||
bool array_range_with_transform() {
|
||||
TEST_START();
|
||||
auto json = R"([1, 2, 3, 4, 5])"_padded;
|
||||
ondemand::parser parser;
|
||||
auto doc = parser.iterate(json);
|
||||
auto arr = doc.get_array();
|
||||
|
||||
auto doubled = ondemand::get_range(arr)
|
||||
| std::views::transform([](auto v) -> int64_t { return int64_t(v) * 2; });
|
||||
|
||||
std::vector<int64_t> values;
|
||||
for (auto val : doubled) {
|
||||
values.push_back(val);
|
||||
}
|
||||
ASSERT_EQUAL(values.size(), size_t(5));
|
||||
ASSERT_EQUAL(values[0], int64_t(2));
|
||||
ASSERT_EQUAL(values[1], int64_t(4));
|
||||
ASSERT_EQUAL(values[2], int64_t(6));
|
||||
ASSERT_EQUAL(values[3], int64_t(8));
|
||||
ASSERT_EQUAL(values[4], int64_t(10));
|
||||
TEST_SUCCEED();
|
||||
}
|
||||
|
||||
bool array_range_strings() {
|
||||
TEST_START();
|
||||
auto json = R"(["alpha", "beta", "gamma"])"_padded;
|
||||
ondemand::parser parser;
|
||||
auto doc = parser.iterate(json);
|
||||
auto arr = doc.get_array();
|
||||
|
||||
auto to_string = [](auto v) -> std::string {
|
||||
return std::string(std::string_view(v));
|
||||
};
|
||||
auto strings = ondemand::get_range(arr) | std::views::transform(to_string);
|
||||
|
||||
std::vector<std::string> values;
|
||||
for (auto s : strings) {
|
||||
values.push_back(s);
|
||||
}
|
||||
ASSERT_EQUAL(values.size(), size_t(3));
|
||||
ASSERT_TRUE(values[0] == "alpha");
|
||||
ASSERT_TRUE(values[1] == "beta");
|
||||
ASSERT_TRUE(values[2] == "gamma");
|
||||
TEST_SUCCEED();
|
||||
}
|
||||
|
||||
bool array_range_empty() {
|
||||
TEST_START();
|
||||
auto json = R"([])"_padded;
|
||||
ondemand::parser parser;
|
||||
auto doc = parser.iterate(json);
|
||||
auto arr = doc.get_array();
|
||||
auto range = ondemand::get_range(arr);
|
||||
|
||||
int count = 0;
|
||||
for (simdjson_unused auto elem : range) {
|
||||
count++;
|
||||
}
|
||||
ASSERT_EQUAL(count, 0);
|
||||
TEST_SUCCEED();
|
||||
}
|
||||
|
||||
bool array_range_nested() {
|
||||
TEST_START();
|
||||
auto json = R"([{"name": "Alice", "age": 30}, {"name": "Bob", "age": 25}])"_padded;
|
||||
ondemand::parser parser;
|
||||
auto doc = parser.iterate(json);
|
||||
auto arr = doc.get_array();
|
||||
|
||||
auto get_name = [](auto v) -> std::string {
|
||||
return std::string(std::string_view(v["name"]));
|
||||
};
|
||||
auto names = ondemand::get_range(arr) | std::views::transform(get_name);
|
||||
|
||||
std::vector<std::string> values;
|
||||
for (auto name : names) {
|
||||
values.push_back(name);
|
||||
}
|
||||
ASSERT_EQUAL(values.size(), size_t(2));
|
||||
ASSERT_TRUE(values[0] == "Alice");
|
||||
ASSERT_TRUE(values[1] == "Bob");
|
||||
TEST_SUCCEED();
|
||||
}
|
||||
|
||||
bool object_get_range_basic() {
|
||||
TEST_START();
|
||||
auto json = R"({"a": 1, "b": 2, "c": 3})"_padded;
|
||||
ondemand::parser parser;
|
||||
auto doc = parser.iterate(json);
|
||||
auto obj = doc.get_object();
|
||||
auto range = ondemand::get_key_value_range(obj);
|
||||
|
||||
std::vector<std::string> keys;
|
||||
std::vector<int64_t> vals;
|
||||
for (auto field_result : range) {
|
||||
keys.push_back(std::string(std::string_view(field_result.escaped_key())));
|
||||
vals.push_back(int64_t(field_result.value()));
|
||||
}
|
||||
ASSERT_EQUAL(keys.size(), size_t(3));
|
||||
ASSERT_TRUE(keys[0] == "a");
|
||||
ASSERT_TRUE(keys[1] == "b");
|
||||
ASSERT_TRUE(keys[2] == "c");
|
||||
ASSERT_EQUAL(vals[0], int64_t(1));
|
||||
ASSERT_EQUAL(vals[1], int64_t(2));
|
||||
ASSERT_EQUAL(vals[2], int64_t(3));
|
||||
TEST_SUCCEED();
|
||||
}
|
||||
|
||||
bool object_range_with_transform() {
|
||||
TEST_START();
|
||||
auto json = R"({"x": 10, "y": 20, "z": 30})"_padded;
|
||||
ondemand::parser parser;
|
||||
auto doc = parser.iterate(json);
|
||||
auto obj = doc.get_object();
|
||||
|
||||
auto get_key = [](auto field_result) -> std::string {
|
||||
return std::string(std::string_view(field_result.escaped_key()));
|
||||
};
|
||||
auto keys = ondemand::get_key_value_range(obj) | std::views::transform(get_key);
|
||||
|
||||
std::vector<std::string> values;
|
||||
for (auto k : keys) {
|
||||
values.push_back(k);
|
||||
}
|
||||
ASSERT_EQUAL(values.size(), size_t(3));
|
||||
ASSERT_TRUE(values[0] == "x");
|
||||
ASSERT_TRUE(values[1] == "y");
|
||||
ASSERT_TRUE(values[2] == "z");
|
||||
TEST_SUCCEED();
|
||||
}
|
||||
|
||||
bool object_range_empty() {
|
||||
TEST_START();
|
||||
auto json = R"({})"_padded;
|
||||
ondemand::parser parser;
|
||||
auto doc = parser.iterate(json);
|
||||
auto obj = doc.get_object();
|
||||
auto range = ondemand::get_key_value_range(obj);
|
||||
|
||||
int count = 0;
|
||||
for (simdjson_unused auto elem : range) {
|
||||
count++;
|
||||
}
|
||||
ASSERT_EQUAL(count, 0);
|
||||
TEST_SUCCEED();
|
||||
}
|
||||
|
||||
bool get_range_from_result() {
|
||||
TEST_START();
|
||||
auto json = R"([100, 200, 300])"_padded;
|
||||
ondemand::parser parser;
|
||||
auto doc = parser.iterate(json);
|
||||
|
||||
// get_range with simdjson_result<array> - unwraps automatically
|
||||
auto range = ondemand::get_range(doc.get_array());
|
||||
|
||||
std::vector<int64_t> values;
|
||||
for (auto elem : range) {
|
||||
values.push_back(int64_t(elem));
|
||||
}
|
||||
ASSERT_EQUAL(values.size(), size_t(3));
|
||||
ASSERT_EQUAL(values[0], int64_t(100));
|
||||
ASSERT_EQUAL(values[1], int64_t(200));
|
||||
ASSERT_EQUAL(values[2], int64_t(300));
|
||||
TEST_SUCCEED();
|
||||
}
|
||||
|
||||
bool object_range_key_iteration() {
|
||||
TEST_START();
|
||||
auto json = R"({"name": "Alice", "age": 30, "city": "New York"})"_padded;
|
||||
ondemand::parser parser;
|
||||
auto doc = parser.iterate(json);
|
||||
auto obj = doc.get_object();
|
||||
|
||||
// Test the specific pattern: iterating over field_result.key()
|
||||
std::vector<std::string> keys;
|
||||
for (auto field_result : ondemand::get_key_value_range(obj)) {
|
||||
keys.push_back(std::string(field_result.escaped_key().value()));
|
||||
}
|
||||
|
||||
ASSERT_EQUAL(keys.size(), size_t(3));
|
||||
bool has_name = false, has_age = false, has_city = false;
|
||||
for (const auto& key : keys) {
|
||||
if (key == "name") has_name = true;
|
||||
else if (key == "age") has_age = true;
|
||||
else if (key == "city") has_city = true;
|
||||
}
|
||||
ASSERT_TRUE(has_name);
|
||||
ASSERT_TRUE(has_age);
|
||||
ASSERT_TRUE(has_city);
|
||||
TEST_SUCCEED();
|
||||
}
|
||||
|
||||
// Verify that the types satisfy the expected C++20 concepts.
|
||||
bool concept_checks() {
|
||||
TEST_START();
|
||||
static_assert(std::input_iterator<ondemand::array_range_iterator>);
|
||||
static_assert(std::input_iterator<ondemand::object_range_iterator>);
|
||||
static_assert(std::ranges::input_range<ondemand::array_range>);
|
||||
static_assert(std::ranges::input_range<ondemand::object_range>);
|
||||
static_assert(std::ranges::view<ondemand::array_range>);
|
||||
static_assert(std::ranges::view<ondemand::object_range>);
|
||||
TEST_SUCCEED();
|
||||
}
|
||||
|
||||
#endif // SIMDJSON_EXCEPTIONS
|
||||
|
||||
// These tests work without exceptions.
|
||||
bool array_range_noexcept_basic() {
|
||||
TEST_START();
|
||||
auto json = R"([1, 2, 3])"_padded;
|
||||
ondemand::parser parser;
|
||||
auto doc = parser.iterate(json);
|
||||
ondemand::array arr;
|
||||
ASSERT_SUCCESS(doc.get_array().get(arr));
|
||||
auto range = ondemand::get_range(arr);
|
||||
ASSERT_SUCCESS(range.error());
|
||||
|
||||
int count = 0;
|
||||
for (auto elem : range) {
|
||||
int64_t val;
|
||||
ASSERT_SUCCESS(elem.get_int64().get(val));
|
||||
count++;
|
||||
}
|
||||
ASSERT_EQUAL(count, 3);
|
||||
TEST_SUCCEED();
|
||||
}
|
||||
|
||||
bool object_range_noexcept_basic() {
|
||||
TEST_START();
|
||||
auto json = R"({"a": 1, "b": 2})"_padded;
|
||||
ondemand::parser parser;
|
||||
auto doc = parser.iterate(json);
|
||||
ondemand::object obj;
|
||||
ASSERT_SUCCESS(doc.get_object().get(obj));
|
||||
auto range = ondemand::get_key_value_range(obj);
|
||||
ASSERT_SUCCESS(range.error());
|
||||
|
||||
int count = 0;
|
||||
for (auto field_result : range) {
|
||||
simdjson_unused ondemand::field f;
|
||||
ASSERT_SUCCESS(std::move(field_result).get(f));
|
||||
count++;
|
||||
}
|
||||
ASSERT_EQUAL(count, 2);
|
||||
TEST_SUCCEED();
|
||||
}
|
||||
|
||||
bool run() {
|
||||
return
|
||||
array_range_noexcept_basic() &&
|
||||
object_range_noexcept_basic() &&
|
||||
#if SIMDJSON_EXCEPTIONS
|
||||
concept_checks() &&
|
||||
array_get_range_basic() &&
|
||||
array_range_with_transform() &&
|
||||
array_range_strings() &&
|
||||
array_range_empty() &&
|
||||
array_range_nested() &&
|
||||
object_get_range_basic() &&
|
||||
object_range_with_transform() &&
|
||||
object_range_empty() &&
|
||||
get_range_from_result() &&
|
||||
object_range_key_iteration() &&
|
||||
#endif // SIMDJSON_EXCEPTIONS
|
||||
true;
|
||||
}
|
||||
|
||||
} // namespace ondemand_ranges_tests
|
||||
|
||||
int main(int argc, char *argv[]) {
|
||||
return test_main(argc, argv, ondemand_ranges_tests::run);
|
||||
}
|
||||
|
||||
#else // !SIMDJSON_SUPPORTS_RANGES
|
||||
|
||||
int main() {
|
||||
std::cout << "Ranges tests require C++20 ranges support, skipping." << std::endl;
|
||||
return 0;
|
||||
}
|
||||
|
||||
#endif // SIMDJSON_SUPPORTS_RANGES
|
||||
Reference in New Issue
Block a user