mirror of
https://github.com/simdjson/simdjson
synced 2026-06-08 17:27:07 +00:00
Compare commits
11 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 920573fffd | |||
| af2a43610e | |||
| e362b18499 | |||
| 6e60ed8338 | |||
| e2ca00c841 | |||
| b7124f7c64 | |||
| e6c6db6493 | |||
| 6fffc99dac | |||
| 8c5cc8c443 | |||
| d8c49a8f25 | |||
| cda98064b8 |
@@ -64,6 +64,7 @@ Real-world usage
|
|||||||
- [RonDB](https://github.com/logicalclocks/rondb)
|
- [RonDB](https://github.com/logicalclocks/rondb)
|
||||||
- [GreptimeDB](https://github.com/GreptimeTeam/greptimedb)
|
- [GreptimeDB](https://github.com/GreptimeTeam/greptimedb)
|
||||||
- [mamba](https://github.com/mamba-org/mamba)
|
- [mamba](https://github.com/mamba-org/mamba)
|
||||||
|
- [Ladybird Browser](https://ladybird.org)
|
||||||
|
|
||||||
|
|
||||||
If you are planning to use simdjson in a product, please work from one of our releases.
|
If you are planning to use simdjson in a product, please work from one of our releases.
|
||||||
|
|||||||
@@ -8,7 +8,7 @@ CPMAddPackage(
|
|||||||
|
|
||||||
option(SIMDJSON_USE_RUST "Build the static_reflect benchmark" OFF)
|
option(SIMDJSON_USE_RUST "Build the static_reflect benchmark" OFF)
|
||||||
|
|
||||||
if(SIMDJSON_USER_RUST)
|
if(SIMDJSON_USE_RUST)
|
||||||
if(NOT WIN32)
|
if(NOT WIN32)
|
||||||
# We want the check whether Rust is available before trying to build a crate.
|
# We want the check whether Rust is available before trying to build a crate.
|
||||||
CPMAddPackage(
|
CPMAddPackage(
|
||||||
@@ -39,9 +39,9 @@ if(SIMDJSON_USER_RUST)
|
|||||||
message(STATUS "curl https://sh.rustup.rs -sSf | sh")
|
message(STATUS "curl https://sh.rustup.rs -sSf | sh")
|
||||||
endif()
|
endif()
|
||||||
endif()
|
endif()
|
||||||
else(SIMDJSON_USER_RUST)
|
else(SIMDJSON_USE_RUST)
|
||||||
message(STATUS "We will not benchmark serde-benchmark." )
|
message(STATUS "We will not benchmark serde-benchmark." )
|
||||||
endif(SIMDJSON_USER_RUST)
|
endif(SIMDJSON_USE_RUST)
|
||||||
|
|
||||||
# Add the benchmark executable targets
|
# Add the benchmark executable targets
|
||||||
add_subdirectory(twitter_benchmark)
|
add_subdirectory(twitter_benchmark)
|
||||||
|
|||||||
@@ -1,4 +1,5 @@
|
|||||||
add_executable(benchmark_serialization_citm_catalog benchmark_serialization_citm_catalog.cpp)
|
add_executable(benchmark_serialization_citm_catalog benchmark_serialization_citm_catalog.cpp)
|
||||||
|
add_executable(benchmark_parsing_citm benchmark_parsing_citm.cpp)
|
||||||
|
|
||||||
# Link with Rust benchmarking code if available
|
# Link with Rust benchmarking code if available
|
||||||
if(TARGET serde-benchmark)
|
if(TARGET serde-benchmark)
|
||||||
@@ -11,4 +12,29 @@ target_link_libraries(benchmark_serialization_citm_catalog PRIVATE simdjson::sim
|
|||||||
target_link_libraries(benchmark_serialization_citm_catalog PRIVATE reflectcpp)
|
target_link_libraries(benchmark_serialization_citm_catalog PRIVATE reflectcpp)
|
||||||
target_compile_definitions(benchmark_serialization_citm_catalog PRIVATE SIMDJSON_BENCH_CPP_REFLECT=1)
|
target_compile_definitions(benchmark_serialization_citm_catalog PRIVATE SIMDJSON_BENCH_CPP_REFLECT=1)
|
||||||
|
|
||||||
target_compile_definitions(benchmark_serialization_citm_catalog PRIVATE JSON_FILE="${BENCH_CITM_JSON}")
|
if(TARGET yyjson)
|
||||||
|
target_link_libraries(benchmark_serialization_citm_catalog PRIVATE yyjson)
|
||||||
|
target_compile_definitions(benchmark_serialization_citm_catalog PRIVATE SIMDJSON_COMPETITION_YYJSON)
|
||||||
|
endif()
|
||||||
|
|
||||||
|
target_compile_definitions(benchmark_serialization_citm_catalog PRIVATE JSON_FILE="${BENCH_CITM_JSON}")
|
||||||
|
|
||||||
|
# Configuration for parsing benchmark
|
||||||
|
if(TARGET serde-benchmark)
|
||||||
|
target_link_libraries(benchmark_parsing_citm PRIVATE serde-benchmark)
|
||||||
|
target_compile_definitions(benchmark_parsing_citm PRIVATE SIMDJSON_RUST_VERSION="${Rust_VERSION}")
|
||||||
|
endif()
|
||||||
|
|
||||||
|
target_link_libraries(benchmark_parsing_citm PRIVATE simdjson::simdjson nlohmann_json)
|
||||||
|
|
||||||
|
if(TARGET rapidjson)
|
||||||
|
target_link_libraries(benchmark_parsing_citm PRIVATE rapidjson)
|
||||||
|
target_compile_definitions(benchmark_parsing_citm PRIVATE SIMDJSON_COMPETITION_RAPIDJSON)
|
||||||
|
endif()
|
||||||
|
|
||||||
|
if(TARGET yyjson)
|
||||||
|
target_link_libraries(benchmark_parsing_citm PRIVATE yyjson)
|
||||||
|
target_compile_definitions(benchmark_parsing_citm PRIVATE SIMDJSON_COMPETITION_YYJSON)
|
||||||
|
endif()
|
||||||
|
|
||||||
|
target_compile_definitions(benchmark_parsing_citm PRIVATE JSON_FILE="${BENCH_CITM_JSON}")
|
||||||
@@ -0,0 +1,564 @@
|
|||||||
|
#include <cassert>
|
||||||
|
#include <cstdlib>
|
||||||
|
#include <ctime>
|
||||||
|
#include <format>
|
||||||
|
#include <fstream>
|
||||||
|
#include <iostream>
|
||||||
|
#include <nlohmann/json.hpp>
|
||||||
|
#include <simdjson.h>
|
||||||
|
#include <string>
|
||||||
|
#include "citm_catalog_data.h"
|
||||||
|
// NOTE: citm_traits.h NOT included because CITM JSON fields are NOT in struct order
|
||||||
|
#include "nlohmann_citm_catalog_data.h"
|
||||||
|
#include "../benchmark_utils/benchmark_helper.h"
|
||||||
|
|
||||||
|
#ifdef SIMDJSON_COMPETITION_RAPIDJSON
|
||||||
|
#include "rapidjson_citm_catalog_data.h"
|
||||||
|
#endif
|
||||||
|
|
||||||
|
#ifdef SIMDJSON_COMPETITION_YYJSON
|
||||||
|
#include "yyjson_citm_catalog_data.h"
|
||||||
|
#endif
|
||||||
|
|
||||||
|
#ifdef SIMDJSON_RUST_VERSION
|
||||||
|
#include "../serde-benchmark/serde_benchmark.h"
|
||||||
|
|
||||||
|
void bench_rust_parsing(const std::string &json_str) {
|
||||||
|
size_t input_volume = json_str.size();
|
||||||
|
printf("# input volume: %zu bytes\n", input_volume);
|
||||||
|
|
||||||
|
volatile bool result = true;
|
||||||
|
pretty_print(1, input_volume, "bench_rust_parsing",
|
||||||
|
bench([&json_str, &result]() {
|
||||||
|
serde_benchmark::CitmCatalog *catalog = serde_benchmark::citm_from_str(json_str.c_str(), json_str.size());
|
||||||
|
result = (catalog != nullptr);
|
||||||
|
if (catalog) {
|
||||||
|
serde_benchmark::free_citm(catalog);
|
||||||
|
}
|
||||||
|
if (!result) {
|
||||||
|
printf("parse error\n");
|
||||||
|
}
|
||||||
|
}));
|
||||||
|
}
|
||||||
|
#endif
|
||||||
|
|
||||||
|
template <class T> void bench_simdjson_static_reflection_parsing(const std::string &json_str) {
|
||||||
|
size_t input_volume = json_str.size();
|
||||||
|
printf("# input volume: %zu bytes\n", input_volume);
|
||||||
|
|
||||||
|
// Pre-allocate padded buffer outside the benchmark loop
|
||||||
|
std::string mutable_json = json_str;
|
||||||
|
simdjson::pad(mutable_json);
|
||||||
|
|
||||||
|
volatile bool result = true;
|
||||||
|
pretty_print(1, input_volume, "bench_simdjson_static_reflection_parsing",
|
||||||
|
bench([&mutable_json, &result]() {
|
||||||
|
simdjson::ondemand::parser parser;
|
||||||
|
simdjson::ondemand::document doc;
|
||||||
|
if(parser.iterate(mutable_json).get(doc)) {
|
||||||
|
result = false;
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
T my_struct;
|
||||||
|
if(doc.get<T>().get(my_struct)) {
|
||||||
|
result = false;
|
||||||
|
}
|
||||||
|
if (!result) {
|
||||||
|
printf("parse error\n");
|
||||||
|
}
|
||||||
|
}));
|
||||||
|
}
|
||||||
|
|
||||||
|
#if SIMDJSON_STATIC_REFLECTION
|
||||||
|
template <class T> void bench_simdjson_from_parsing(const std::string &json_str) {
|
||||||
|
size_t input_volume = json_str.size();
|
||||||
|
printf("# input volume: %zu bytes\n", input_volume);
|
||||||
|
|
||||||
|
// Pre-allocate padded buffer outside the benchmark loop
|
||||||
|
simdjson::padded_string padded = simdjson::padded_string(json_str);
|
||||||
|
|
||||||
|
volatile bool result = true;
|
||||||
|
pretty_print(1, input_volume, "bench_simdjson_from_parsing",
|
||||||
|
bench([&padded, &result]() {
|
||||||
|
T my_struct;
|
||||||
|
auto err = simdjson::from(padded).get(my_struct);
|
||||||
|
if (err) {
|
||||||
|
result = false;
|
||||||
|
printf("parse error: %s\n", simdjson::error_message(err));
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
}));
|
||||||
|
}
|
||||||
|
#endif
|
||||||
|
|
||||||
|
// nlohmann::json deserialization functions
|
||||||
|
void from_json(const nlohmann::json &j, CITMPrice &p) {
|
||||||
|
j.at("amount").get_to(p.amount);
|
||||||
|
j.at("audienceSubCategoryId").get_to(p.audienceSubCategoryId);
|
||||||
|
j.at("seatCategoryId").get_to(p.seatCategoryId);
|
||||||
|
}
|
||||||
|
|
||||||
|
void from_json(const nlohmann::json &j, CITMArea &a) {
|
||||||
|
j.at("areaId").get_to(a.areaId);
|
||||||
|
j.at("blockIds").get_to(a.blockIds);
|
||||||
|
}
|
||||||
|
|
||||||
|
void from_json(const nlohmann::json &j, CITMSeatCategory &s) {
|
||||||
|
j.at("areas").get_to(s.areas);
|
||||||
|
j.at("seatCategoryId").get_to(s.seatCategoryId);
|
||||||
|
}
|
||||||
|
|
||||||
|
void from_json(const nlohmann::json &j, CITMPerformance &p) {
|
||||||
|
j.at("id").get_to(p.id);
|
||||||
|
j.at("eventId").get_to(p.eventId);
|
||||||
|
if (j.contains("logo") && !j["logo"].is_null()) {
|
||||||
|
p.logo = j["logo"].get<std::string>();
|
||||||
|
}
|
||||||
|
if (j.contains("name") && !j["name"].is_null()) {
|
||||||
|
p.name = j["name"].get<std::string>();
|
||||||
|
}
|
||||||
|
j.at("prices").get_to(p.prices);
|
||||||
|
j.at("seatCategories").get_to(p.seatCategories);
|
||||||
|
if (j.contains("seatMapImage") && !j["seatMapImage"].is_null()) {
|
||||||
|
p.seatMapImage = j["seatMapImage"].get<std::string>();
|
||||||
|
}
|
||||||
|
j.at("start").get_to(p.start);
|
||||||
|
j.at("venueCode").get_to(p.venueCode);
|
||||||
|
}
|
||||||
|
|
||||||
|
void from_json(const nlohmann::json &j, CITMEvent &e) {
|
||||||
|
j.at("id").get_to(e.id);
|
||||||
|
j.at("name").get_to(e.name);
|
||||||
|
if (j.contains("description") && !j["description"].is_null()) {
|
||||||
|
e.description = j["description"].get<std::string>();
|
||||||
|
}
|
||||||
|
if (j.contains("logo") && !j["logo"].is_null()) {
|
||||||
|
e.logo = j["logo"].get<std::string>();
|
||||||
|
}
|
||||||
|
j.at("subTopicIds").get_to(e.subTopicIds);
|
||||||
|
if (j.contains("subjectCode") && !j["subjectCode"].is_null()) {
|
||||||
|
e.subjectCode = j["subjectCode"].get<std::string>();
|
||||||
|
}
|
||||||
|
if (j.contains("subtitle") && !j["subtitle"].is_null()) {
|
||||||
|
e.subtitle = j["subtitle"].get<std::string>();
|
||||||
|
}
|
||||||
|
j.at("topicIds").get_to(e.topicIds);
|
||||||
|
}
|
||||||
|
|
||||||
|
void from_json(const nlohmann::json &j, CitmCatalog &c) {
|
||||||
|
j.at("events").get_to(c.events);
|
||||||
|
j.at("performances").get_to(c.performances);
|
||||||
|
}
|
||||||
|
|
||||||
|
CitmCatalog nlohmann_deserialize(const std::string &json_str) {
|
||||||
|
nlohmann::json j = nlohmann::json::parse(json_str);
|
||||||
|
return j.get<CitmCatalog>();
|
||||||
|
}
|
||||||
|
|
||||||
|
void bench_nlohmann_parsing(const std::string &json_str) {
|
||||||
|
size_t input_volume = json_str.size();
|
||||||
|
printf("# input volume: %zu bytes\n", input_volume);
|
||||||
|
|
||||||
|
volatile bool result = true;
|
||||||
|
pretty_print(1, input_volume, "bench_nlohmann_parsing",
|
||||||
|
bench([&json_str, &result]() {
|
||||||
|
try {
|
||||||
|
CitmCatalog data = nlohmann_deserialize(json_str);
|
||||||
|
result = true;
|
||||||
|
} catch (...) {
|
||||||
|
result = false;
|
||||||
|
printf("parse error\n");
|
||||||
|
}
|
||||||
|
}));
|
||||||
|
}
|
||||||
|
|
||||||
|
#ifdef SIMDJSON_COMPETITION_RAPIDJSON
|
||||||
|
CitmCatalog rapidjson_deserialize(const std::string &json_str) {
|
||||||
|
rapidjson::Document doc;
|
||||||
|
doc.Parse(json_str.c_str());
|
||||||
|
|
||||||
|
if (doc.HasParseError()) {
|
||||||
|
throw std::runtime_error("RapidJSON parse error");
|
||||||
|
}
|
||||||
|
|
||||||
|
CitmCatalog catalog;
|
||||||
|
|
||||||
|
// Parse events
|
||||||
|
if (doc.HasMember("events") && doc["events"].IsObject()) {
|
||||||
|
for (auto& m : doc["events"].GetObject()) {
|
||||||
|
CITMEvent event;
|
||||||
|
const auto& e = m.value;
|
||||||
|
|
||||||
|
event.id = e["id"].GetUint64();
|
||||||
|
event.name = e["name"].GetString();
|
||||||
|
if (e.HasMember("description") && !e["description"].IsNull()) {
|
||||||
|
event.description = e["description"].GetString();
|
||||||
|
}
|
||||||
|
if (e.HasMember("logo") && !e["logo"].IsNull()) {
|
||||||
|
event.logo = e["logo"].GetString();
|
||||||
|
}
|
||||||
|
|
||||||
|
event.subTopicIds.clear();
|
||||||
|
for (auto& id : e["subTopicIds"].GetArray()) {
|
||||||
|
event.subTopicIds.push_back(id.GetUint64());
|
||||||
|
}
|
||||||
|
|
||||||
|
if (e.HasMember("subjectCode") && !e["subjectCode"].IsNull()) {
|
||||||
|
event.subjectCode = e["subjectCode"].GetString();
|
||||||
|
}
|
||||||
|
if (e.HasMember("subtitle") && !e["subtitle"].IsNull()) {
|
||||||
|
event.subtitle = e["subtitle"].GetString();
|
||||||
|
}
|
||||||
|
|
||||||
|
event.topicIds.clear();
|
||||||
|
for (auto& id : e["topicIds"].GetArray()) {
|
||||||
|
event.topicIds.push_back(id.GetUint64());
|
||||||
|
}
|
||||||
|
|
||||||
|
catalog.events[m.name.GetString()] = event;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Parse performances
|
||||||
|
if (doc.HasMember("performances") && doc["performances"].IsArray()) {
|
||||||
|
for (auto& p : doc["performances"].GetArray()) {
|
||||||
|
CITMPerformance perf;
|
||||||
|
|
||||||
|
perf.id = p["id"].GetUint64();
|
||||||
|
perf.eventId = p["eventId"].GetUint64();
|
||||||
|
if (p.HasMember("logo") && !p["logo"].IsNull()) {
|
||||||
|
perf.logo = p["logo"].GetString();
|
||||||
|
}
|
||||||
|
if (p.HasMember("name") && !p["name"].IsNull()) {
|
||||||
|
perf.name = p["name"].GetString();
|
||||||
|
}
|
||||||
|
|
||||||
|
// Parse prices
|
||||||
|
for (auto& price : p["prices"].GetArray()) {
|
||||||
|
CITMPrice pr;
|
||||||
|
pr.amount = price["amount"].GetUint64();
|
||||||
|
pr.audienceSubCategoryId = price["audienceSubCategoryId"].GetUint64();
|
||||||
|
pr.seatCategoryId = price["seatCategoryId"].GetUint64();
|
||||||
|
perf.prices.push_back(pr);
|
||||||
|
}
|
||||||
|
|
||||||
|
// Parse seat categories
|
||||||
|
for (auto& sc : p["seatCategories"].GetArray()) {
|
||||||
|
CITMSeatCategory seatCat;
|
||||||
|
seatCat.seatCategoryId = sc["seatCategoryId"].GetUint64();
|
||||||
|
|
||||||
|
for (auto& area : sc["areas"].GetArray()) {
|
||||||
|
CITMArea ar;
|
||||||
|
ar.areaId = area["areaId"].GetUint64();
|
||||||
|
for (auto& block : area["blockIds"].GetArray()) {
|
||||||
|
ar.blockIds.push_back(block.GetUint64());
|
||||||
|
}
|
||||||
|
seatCat.areas.push_back(ar);
|
||||||
|
}
|
||||||
|
perf.seatCategories.push_back(seatCat);
|
||||||
|
}
|
||||||
|
|
||||||
|
if (p.HasMember("seatMapImage") && !p["seatMapImage"].IsNull()) {
|
||||||
|
perf.seatMapImage = p["seatMapImage"].GetString();
|
||||||
|
}
|
||||||
|
perf.start = p["start"].GetUint64();
|
||||||
|
perf.venueCode = p["venueCode"].GetString();
|
||||||
|
|
||||||
|
catalog.performances.push_back(perf);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
return catalog;
|
||||||
|
}
|
||||||
|
|
||||||
|
void bench_rapidjson_parsing(const std::string &json_str) {
|
||||||
|
size_t input_volume = json_str.size();
|
||||||
|
printf("# input volume: %zu bytes\n", input_volume);
|
||||||
|
|
||||||
|
volatile bool result = true;
|
||||||
|
pretty_print(1, input_volume, "bench_rapidjson_parsing",
|
||||||
|
bench([&json_str, &result]() {
|
||||||
|
try {
|
||||||
|
CitmCatalog data = rapidjson_deserialize(json_str);
|
||||||
|
result = true;
|
||||||
|
} catch (...) {
|
||||||
|
result = false;
|
||||||
|
printf("parse error\n");
|
||||||
|
}
|
||||||
|
}));
|
||||||
|
}
|
||||||
|
#endif
|
||||||
|
|
||||||
|
#ifdef SIMDJSON_COMPETITION_YYJSON
|
||||||
|
CitmCatalog yyjson_deserialize(const std::string &json_str) {
|
||||||
|
yyjson_doc *doc = yyjson_read(json_str.c_str(), json_str.size(), 0);
|
||||||
|
if (!doc) {
|
||||||
|
throw std::runtime_error("YYJson parse error");
|
||||||
|
}
|
||||||
|
|
||||||
|
yyjson_val *root = yyjson_doc_get_root(doc);
|
||||||
|
CitmCatalog catalog;
|
||||||
|
|
||||||
|
// Parse events
|
||||||
|
yyjson_val *events = yyjson_obj_get(root, "events");
|
||||||
|
if (events) {
|
||||||
|
size_t idx, max;
|
||||||
|
yyjson_val *key, *val;
|
||||||
|
yyjson_obj_foreach(events, idx, max, key, val) {
|
||||||
|
CITMEvent event;
|
||||||
|
|
||||||
|
event.id = yyjson_get_uint(yyjson_obj_get(val, "id"));
|
||||||
|
const char* name = yyjson_get_str(yyjson_obj_get(val, "name"));
|
||||||
|
if (name) event.name = name;
|
||||||
|
|
||||||
|
yyjson_val *desc = yyjson_obj_get(val, "description");
|
||||||
|
if (desc && !yyjson_is_null(desc)) {
|
||||||
|
const char* str = yyjson_get_str(desc);
|
||||||
|
if (str) event.description = str;
|
||||||
|
}
|
||||||
|
|
||||||
|
yyjson_val *logo = yyjson_obj_get(val, "logo");
|
||||||
|
if (logo && !yyjson_is_null(logo)) {
|
||||||
|
const char* str = yyjson_get_str(logo);
|
||||||
|
if (str) event.logo = str;
|
||||||
|
}
|
||||||
|
|
||||||
|
yyjson_val *subTopics = yyjson_obj_get(val, "subTopicIds");
|
||||||
|
if (subTopics) {
|
||||||
|
size_t sidx, smax;
|
||||||
|
yyjson_val *sval;
|
||||||
|
yyjson_arr_foreach(subTopics, sidx, smax, sval) {
|
||||||
|
event.subTopicIds.push_back(yyjson_get_uint(sval));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
yyjson_val *subjectCode = yyjson_obj_get(val, "subjectCode");
|
||||||
|
if (subjectCode && !yyjson_is_null(subjectCode)) {
|
||||||
|
const char* str = yyjson_get_str(subjectCode);
|
||||||
|
if (str) event.subjectCode = str;
|
||||||
|
}
|
||||||
|
|
||||||
|
yyjson_val *subtitle = yyjson_obj_get(val, "subtitle");
|
||||||
|
if (subtitle && !yyjson_is_null(subtitle)) {
|
||||||
|
const char* str = yyjson_get_str(subtitle);
|
||||||
|
if (str) event.subtitle = str;
|
||||||
|
}
|
||||||
|
|
||||||
|
yyjson_val *topics = yyjson_obj_get(val, "topicIds");
|
||||||
|
if (topics) {
|
||||||
|
size_t tidx, tmax;
|
||||||
|
yyjson_val *tval;
|
||||||
|
yyjson_arr_foreach(topics, tidx, tmax, tval) {
|
||||||
|
event.topicIds.push_back(yyjson_get_uint(tval));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
const char* keyStr = yyjson_get_str(key);
|
||||||
|
if (keyStr) {
|
||||||
|
catalog.events[keyStr] = event;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Parse performances
|
||||||
|
yyjson_val *performances = yyjson_obj_get(root, "performances");
|
||||||
|
if (performances) {
|
||||||
|
size_t idx, max;
|
||||||
|
yyjson_val *val;
|
||||||
|
yyjson_arr_foreach(performances, idx, max, val) {
|
||||||
|
CITMPerformance perf;
|
||||||
|
|
||||||
|
perf.id = yyjson_get_uint(yyjson_obj_get(val, "id"));
|
||||||
|
perf.eventId = yyjson_get_uint(yyjson_obj_get(val, "eventId"));
|
||||||
|
|
||||||
|
yyjson_val *logo = yyjson_obj_get(val, "logo");
|
||||||
|
if (logo && !yyjson_is_null(logo)) {
|
||||||
|
const char* str = yyjson_get_str(logo);
|
||||||
|
if (str) perf.logo = str;
|
||||||
|
}
|
||||||
|
|
||||||
|
yyjson_val *name = yyjson_obj_get(val, "name");
|
||||||
|
if (name && !yyjson_is_null(name)) {
|
||||||
|
const char* str = yyjson_get_str(name);
|
||||||
|
if (str) perf.name = str;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Parse prices
|
||||||
|
yyjson_val *prices = yyjson_obj_get(val, "prices");
|
||||||
|
if (prices) {
|
||||||
|
size_t pidx, pmax;
|
||||||
|
yyjson_val *pval;
|
||||||
|
yyjson_arr_foreach(prices, pidx, pmax, pval) {
|
||||||
|
CITMPrice price;
|
||||||
|
price.amount = yyjson_get_uint(yyjson_obj_get(pval, "amount"));
|
||||||
|
price.audienceSubCategoryId = yyjson_get_uint(yyjson_obj_get(pval, "audienceSubCategoryId"));
|
||||||
|
price.seatCategoryId = yyjson_get_uint(yyjson_obj_get(pval, "seatCategoryId"));
|
||||||
|
perf.prices.push_back(price);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Parse seat categories
|
||||||
|
yyjson_val *seatCats = yyjson_obj_get(val, "seatCategories");
|
||||||
|
if (seatCats) {
|
||||||
|
size_t scidx, scmax;
|
||||||
|
yyjson_val *scval;
|
||||||
|
yyjson_arr_foreach(seatCats, scidx, scmax, scval) {
|
||||||
|
CITMSeatCategory seatCat;
|
||||||
|
seatCat.seatCategoryId = yyjson_get_uint(yyjson_obj_get(scval, "seatCategoryId"));
|
||||||
|
|
||||||
|
yyjson_val *areas = yyjson_obj_get(scval, "areas");
|
||||||
|
if (areas) {
|
||||||
|
size_t aidx, amax;
|
||||||
|
yyjson_val *aval;
|
||||||
|
yyjson_arr_foreach(areas, aidx, amax, aval) {
|
||||||
|
CITMArea area;
|
||||||
|
area.areaId = yyjson_get_uint(yyjson_obj_get(aval, "areaId"));
|
||||||
|
|
||||||
|
yyjson_val *blocks = yyjson_obj_get(aval, "blockIds");
|
||||||
|
if (blocks) {
|
||||||
|
size_t bidx, bmax;
|
||||||
|
yyjson_val *bval;
|
||||||
|
yyjson_arr_foreach(blocks, bidx, bmax, bval) {
|
||||||
|
area.blockIds.push_back(yyjson_get_uint(bval));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
seatCat.areas.push_back(area);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
perf.seatCategories.push_back(seatCat);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
yyjson_val *seatMapImage = yyjson_obj_get(val, "seatMapImage");
|
||||||
|
if (seatMapImage && !yyjson_is_null(seatMapImage)) {
|
||||||
|
const char* str = yyjson_get_str(seatMapImage);
|
||||||
|
if (str) perf.seatMapImage = str;
|
||||||
|
}
|
||||||
|
|
||||||
|
perf.start = yyjson_get_uint(yyjson_obj_get(val, "start"));
|
||||||
|
const char* venueCode = yyjson_get_str(yyjson_obj_get(val, "venueCode"));
|
||||||
|
if (venueCode) perf.venueCode = venueCode;
|
||||||
|
|
||||||
|
catalog.performances.push_back(perf);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
yyjson_doc_free(doc);
|
||||||
|
return catalog;
|
||||||
|
}
|
||||||
|
|
||||||
|
void bench_yyjson_parsing(const std::string &json_str) {
|
||||||
|
size_t input_volume = json_str.size();
|
||||||
|
printf("# input volume: %zu bytes\n", input_volume);
|
||||||
|
|
||||||
|
volatile bool result = true;
|
||||||
|
pretty_print(1, input_volume, "bench_yyjson_parsing",
|
||||||
|
bench([&json_str, &result]() {
|
||||||
|
try {
|
||||||
|
CitmCatalog data = yyjson_deserialize(json_str);
|
||||||
|
result = true;
|
||||||
|
} catch (...) {
|
||||||
|
result = false;
|
||||||
|
printf("parse error\n");
|
||||||
|
}
|
||||||
|
}));
|
||||||
|
}
|
||||||
|
#endif
|
||||||
|
|
||||||
|
std::string read_file(std::string filename) {
|
||||||
|
printf("# Reading file %s\n", filename.c_str());
|
||||||
|
constexpr size_t read_size = 4096;
|
||||||
|
auto stream = std::ifstream(filename);
|
||||||
|
stream.exceptions(std::ios_base::badbit);
|
||||||
|
|
||||||
|
if (!stream) {
|
||||||
|
std::cerr << "Error: Failed to open file " << filename << std::endl;
|
||||||
|
exit(EXIT_FAILURE);
|
||||||
|
}
|
||||||
|
|
||||||
|
std::string out;
|
||||||
|
auto buf = std::string(read_size, '\0');
|
||||||
|
while (stream.read(&buf[0], read_size)) {
|
||||||
|
out.append(buf, 0, size_t(stream.gcount()));
|
||||||
|
}
|
||||||
|
out.append(buf, 0, size_t(stream.gcount()));
|
||||||
|
return out;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Function to check if benchmark name matches any of the comma-separated filters
|
||||||
|
bool matches_filter(const std::string& benchmark_name, const std::string& filter) {
|
||||||
|
if (filter.empty()) return true;
|
||||||
|
|
||||||
|
// Split filter by comma
|
||||||
|
size_t start = 0;
|
||||||
|
size_t end = filter.find(',');
|
||||||
|
while (end != std::string::npos) {
|
||||||
|
std::string token = filter.substr(start, end - start);
|
||||||
|
if (benchmark_name.find(token) != std::string::npos) {
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
start = end + 1;
|
||||||
|
end = filter.find(',', start);
|
||||||
|
}
|
||||||
|
// Check last token
|
||||||
|
std::string token = filter.substr(start);
|
||||||
|
return benchmark_name.find(token) != std::string::npos;
|
||||||
|
}
|
||||||
|
|
||||||
|
int main(int argc, char *argv[]) {
|
||||||
|
// Get the JSON file path from preprocessor or use default
|
||||||
|
std::string filename;
|
||||||
|
#ifdef JSON_FILE
|
||||||
|
filename = JSON_FILE;
|
||||||
|
#else
|
||||||
|
filename = "jsonexamples/citm_catalog.json";
|
||||||
|
#endif
|
||||||
|
|
||||||
|
std::string json_str = read_file(filename);
|
||||||
|
|
||||||
|
// Parse command-line arguments for filter
|
||||||
|
std::string filter;
|
||||||
|
for (int i = 1; i < argc; i++) {
|
||||||
|
std::string arg = argv[i];
|
||||||
|
if (arg == "-f" && i + 1 < argc) {
|
||||||
|
filter = argv[i + 1];
|
||||||
|
printf("# Filter: %s\n", filter.c_str());
|
||||||
|
i++;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// If no filter provided, run all benchmarks
|
||||||
|
if (filter.empty()) {
|
||||||
|
printf("# Running all benchmarks (use -f <filter> to run specific ones)\n");
|
||||||
|
}
|
||||||
|
|
||||||
|
// Benchmarking the parsing
|
||||||
|
if (matches_filter("nlohmann", filter)) {
|
||||||
|
bench_nlohmann_parsing(json_str);
|
||||||
|
}
|
||||||
|
#ifdef SIMDJSON_COMPETITION_RAPIDJSON
|
||||||
|
if (matches_filter("rapidjson", filter)) {
|
||||||
|
bench_rapidjson_parsing(json_str);
|
||||||
|
}
|
||||||
|
#endif
|
||||||
|
#ifdef SIMDJSON_COMPETITION_YYJSON
|
||||||
|
if (matches_filter("yyjson", filter)) {
|
||||||
|
bench_yyjson_parsing(json_str);
|
||||||
|
}
|
||||||
|
#endif
|
||||||
|
if (matches_filter("simdjson_static_reflection", filter)) {
|
||||||
|
bench_simdjson_static_reflection_parsing<CitmCatalog>(json_str);
|
||||||
|
}
|
||||||
|
#if SIMDJSON_STATIC_REFLECTION
|
||||||
|
if (matches_filter("simdjson_from", filter)) {
|
||||||
|
bench_simdjson_from_parsing<CitmCatalog>(json_str);
|
||||||
|
}
|
||||||
|
#endif
|
||||||
|
#ifdef SIMDJSON_RUST_VERSION
|
||||||
|
if (matches_filter("rust", filter)) {
|
||||||
|
printf("# Note: Rust/Serde parsing test\n");
|
||||||
|
bench_rust_parsing(json_str);
|
||||||
|
}
|
||||||
|
#endif
|
||||||
|
return EXIT_SUCCESS;
|
||||||
|
}
|
||||||
+160
-22
@@ -11,6 +11,10 @@
|
|||||||
#include "nlohmann_citm_catalog_data.h"
|
#include "nlohmann_citm_catalog_data.h"
|
||||||
#include "../benchmark_utils/benchmark_helper.h"
|
#include "../benchmark_utils/benchmark_helper.h"
|
||||||
|
|
||||||
|
#ifdef SIMDJSON_COMPETITION_YYJSON
|
||||||
|
#include "yyjson_citm_catalog_data.h"
|
||||||
|
#endif
|
||||||
|
|
||||||
#if SIMDJSON_BENCH_CPP_REFLECT
|
#if SIMDJSON_BENCH_CPP_REFLECT
|
||||||
#include <rfl.hpp>
|
#include <rfl.hpp>
|
||||||
#include <rfl/json.hpp>
|
#include <rfl/json.hpp>
|
||||||
@@ -35,20 +39,15 @@ void bench_reflect_cpp(CitmCatalog &data) {
|
|||||||
#include "../serde-benchmark/serde_benchmark.h"
|
#include "../serde-benchmark/serde_benchmark.h"
|
||||||
|
|
||||||
void bench_rust(serde_benchmark::CitmCatalog *data) {
|
void bench_rust(serde_benchmark::CitmCatalog *data) {
|
||||||
const char * output = serde_benchmark::str_from_citm(data);
|
serde_benchmark::set_citm_data(data);
|
||||||
size_t output_volume = strlen(output);
|
size_t output_volume = serde_benchmark::serialize_citm_to_string();
|
||||||
printf("# output volume: %zu bytes\n", output_volume);
|
printf("# output volume: %zu bytes\n", output_volume);
|
||||||
|
|
||||||
volatile size_t measured_volume = 0;
|
volatile size_t measured_volume = 0;
|
||||||
pretty_print(1, output_volume, "bench_rust",
|
pretty_print(1, output_volume, "bench_rust",
|
||||||
bench([&data, &measured_volume, &output_volume]() {
|
bench([&measured_volume, &output_volume]() {
|
||||||
const char * output = serde_benchmark::str_from_citm(data);
|
measured_volume = serde_benchmark::serialize_citm_to_string();
|
||||||
measured_volume = strlen(output);
|
|
||||||
if (measured_volume != output_volume) {
|
|
||||||
printf("mismatch\n");
|
|
||||||
}
|
|
||||||
serde_benchmark::free_str(const_cast<char*>(output));
|
|
||||||
}));
|
}));
|
||||||
serde_benchmark::free_str(const_cast<char*>(output));
|
|
||||||
}
|
}
|
||||||
#endif // SIMDJSON_RUST_VERSION
|
#endif // SIMDJSON_RUST_VERSION
|
||||||
|
|
||||||
@@ -68,7 +67,55 @@ void bench_nlohmann(CitmCatalog &data) {
|
|||||||
}));
|
}));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#ifdef SIMDJSON_COMPETITION_YYJSON
|
||||||
|
void bench_yyjson(CitmCatalog &data) {
|
||||||
|
std::string output = yyjson_serialize_citm(data);
|
||||||
|
size_t output_volume = output.size();
|
||||||
|
printf("# output volume: %zu bytes\n", output_volume);
|
||||||
|
|
||||||
|
volatile size_t measured_volume = 0;
|
||||||
|
pretty_print(1, output_volume, "bench_yyjson",
|
||||||
|
bench([&data, &measured_volume, &output_volume]() {
|
||||||
|
std::string output = yyjson_serialize_citm(data);
|
||||||
|
measured_volume = output.size();
|
||||||
|
if (measured_volume != output_volume) {
|
||||||
|
printf("mismatch\n");
|
||||||
|
}
|
||||||
|
}));
|
||||||
|
}
|
||||||
|
#endif
|
||||||
|
|
||||||
|
// Fair allocation variant: allocates fresh buffer each iteration (matches other libraries)
|
||||||
void bench_simdjson_static_reflection(CitmCatalog &data) {
|
void bench_simdjson_static_reflection(CitmCatalog &data) {
|
||||||
|
// First run to determine expected size
|
||||||
|
simdjson::builder::string_builder sb_init;
|
||||||
|
simdjson::builder::append(sb_init, data);
|
||||||
|
std::string_view p_init;
|
||||||
|
if(sb_init.view().get(p_init)) {
|
||||||
|
std::cerr << "Error!" << std::endl;
|
||||||
|
}
|
||||||
|
size_t output_volume = p_init.size();
|
||||||
|
printf("# output volume: %zu bytes\n", output_volume);
|
||||||
|
|
||||||
|
volatile size_t measured_volume = 0;
|
||||||
|
pretty_print(sizeof(data), output_volume, "bench_simdjson_static_reflection",
|
||||||
|
bench([&data, &measured_volume, &output_volume]() {
|
||||||
|
// Fresh allocation each iteration - fair comparison
|
||||||
|
simdjson::builder::string_builder sb;
|
||||||
|
simdjson::builder::append(sb, data);
|
||||||
|
std::string_view p;
|
||||||
|
if(sb.view().get(p)) {
|
||||||
|
std::cerr << "Error!" << std::endl;
|
||||||
|
}
|
||||||
|
measured_volume = sb.size();
|
||||||
|
if (measured_volume != output_volume) {
|
||||||
|
printf("mismatch\n");
|
||||||
|
}
|
||||||
|
}));
|
||||||
|
}
|
||||||
|
|
||||||
|
// Optimized variant: reuses buffer across iterations (shows API potential)
|
||||||
|
void bench_simdjson_static_reflection_reuse(CitmCatalog &data) {
|
||||||
simdjson::builder::string_builder sb;
|
simdjson::builder::string_builder sb;
|
||||||
simdjson::builder::append(sb, data);
|
simdjson::builder::append(sb, data);
|
||||||
std::string_view p;
|
std::string_view p;
|
||||||
@@ -80,7 +127,7 @@ void bench_simdjson_static_reflection(CitmCatalog &data) {
|
|||||||
printf("# output volume: %zu bytes\n", output_volume);
|
printf("# output volume: %zu bytes\n", output_volume);
|
||||||
|
|
||||||
volatile size_t measured_volume = 0;
|
volatile size_t measured_volume = 0;
|
||||||
pretty_print(sizeof(data), output_volume, "bench_simdjson_static_reflection",
|
pretty_print(sizeof(data), output_volume, "bench_simdjson_reuse_buffer",
|
||||||
bench([&data, &measured_volume, &output_volume, &sb]() {
|
bench([&data, &measured_volume, &output_volume, &sb]() {
|
||||||
sb.clear();
|
sb.clear();
|
||||||
simdjson::builder::append(sb, data);
|
simdjson::builder::append(sb, data);
|
||||||
@@ -95,25 +142,97 @@ void bench_simdjson_static_reflection(CitmCatalog &data) {
|
|||||||
}));
|
}));
|
||||||
}
|
}
|
||||||
|
|
||||||
std::string read_file(const std::string &file_path, size_t read_size = 65536) {
|
#if SIMDJSON_STATIC_REFLECTION
|
||||||
|
// Fair allocation variant: allocates fresh string each iteration
|
||||||
|
void bench_simdjson_to(CitmCatalog &data) {
|
||||||
|
// First run to determine size
|
||||||
|
std::string output_init;
|
||||||
|
if (simdjson::error_code err = simdjson::builder::to_json(data, output_init); err) {
|
||||||
|
std::cerr << "Error in to_json initialization!" << simdjson::error_message(err) << std::endl;
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
size_t output_volume = output_init.size();
|
||||||
|
printf("# output volume: %zu bytes\n", output_volume);
|
||||||
|
|
||||||
|
volatile size_t measured_volume = 0;
|
||||||
|
pretty_print(sizeof(data), output_volume, "bench_simdjson_to",
|
||||||
|
bench([&data, &measured_volume, &output_volume]() {
|
||||||
|
// Fresh allocation each iteration - fair comparison
|
||||||
|
std::string output;
|
||||||
|
if (simdjson::error_code err = simdjson::builder::to_json(data, output); err) {
|
||||||
|
std::cerr << "Error in to_json!" << simdjson::error_message(err) << std::endl;
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
measured_volume = output.size();
|
||||||
|
if (measured_volume != output_volume) {
|
||||||
|
printf("mismatch\n");
|
||||||
|
}
|
||||||
|
}));
|
||||||
|
}
|
||||||
|
|
||||||
|
// Optimized variant: reuses pre-allocated string
|
||||||
|
void bench_simdjson_to_reuse(CitmCatalog &data) {
|
||||||
|
std::string output;
|
||||||
|
if (simdjson::error_code err = simdjson::builder::to_json(data, output); err) {
|
||||||
|
std::cerr << "Error in to_json initialization!" << simdjson::error_message(err) << std::endl;
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
size_t output_volume = output.size();
|
||||||
|
printf("# output volume: %zu bytes\n", output_volume);
|
||||||
|
|
||||||
|
// Pre-allocate string with sufficient capacity to avoid reallocation
|
||||||
|
output.reserve(output_volume * 2);
|
||||||
|
|
||||||
|
volatile size_t measured_volume = 0;
|
||||||
|
pretty_print(sizeof(data), output_volume, "bench_simdjson_to_reuse",
|
||||||
|
bench([&data, &measured_volume, &output_volume, &output]() {
|
||||||
|
// Reuse the pre-allocated string - avoids allocation
|
||||||
|
if (simdjson::error_code err = simdjson::builder::to_json(data, output); err) {
|
||||||
|
std::cerr << "Error in to_json!" << simdjson::error_message(err) << std::endl;
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
measured_volume = output.size();
|
||||||
|
if (measured_volume != output_volume) {
|
||||||
|
printf("mismatch\n");
|
||||||
|
}
|
||||||
|
}));
|
||||||
|
}
|
||||||
|
#endif
|
||||||
|
|
||||||
|
simdjson::padded_string read_file(const std::string &file_path, size_t read_size = 65536) {
|
||||||
std::ifstream stream(file_path, std::ios::binary);
|
std::ifstream stream(file_path, std::ios::binary);
|
||||||
if(!stream) {
|
if(!stream) {
|
||||||
std::cerr << "Could not open file '" << file_path << "'" << std::endl;
|
std::cerr << "Could not open file '" << file_path << "'" << std::endl;
|
||||||
exit(EXIT_FAILURE);
|
exit(EXIT_FAILURE);
|
||||||
}
|
}
|
||||||
stream.exceptions(std::ios_base::badbit);
|
stream.exceptions(std::ios_base::badbit);
|
||||||
std::string out;
|
simdjson::padded_string_builder builder;
|
||||||
std::string buf(read_size, '\0');
|
std::string buf(read_size, '\0');
|
||||||
while (stream.read(&buf[0], read_size)) {
|
while (stream.read(&buf[0], read_size)) {
|
||||||
out.append(buf, 0, size_t(stream.gcount()));
|
builder.append(buf.data(), size_t(stream.gcount()));
|
||||||
}
|
}
|
||||||
out.append(buf, 0, size_t(stream.gcount()));
|
builder.append(buf.data(), size_t(stream.gcount()));
|
||||||
return out;
|
return builder.convert();
|
||||||
}
|
}
|
||||||
|
|
||||||
// Function to check if benchmark name contains filter substring
|
// Function to check if benchmark name matches any of the comma-separated filters
|
||||||
bool matches_filter(const std::string& benchmark_name, const std::string& filter) {
|
bool matches_filter(const std::string& benchmark_name, const std::string& filter) {
|
||||||
return filter.empty() || benchmark_name.find(filter) != std::string::npos;
|
if (filter.empty()) return true;
|
||||||
|
|
||||||
|
// Split filter by comma
|
||||||
|
size_t start = 0;
|
||||||
|
size_t end = filter.find(',');
|
||||||
|
while (end != std::string::npos) {
|
||||||
|
std::string token = filter.substr(start, end - start);
|
||||||
|
if (benchmark_name.find(token) != std::string::npos) {
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
start = end + 1;
|
||||||
|
end = filter.find(',', start);
|
||||||
|
}
|
||||||
|
// Check last token
|
||||||
|
std::string token = filter.substr(start);
|
||||||
|
return benchmark_name.find(token) != std::string::npos;
|
||||||
}
|
}
|
||||||
|
|
||||||
int main(int argc, char* argv[]) {
|
int main(int argc, char* argv[]) {
|
||||||
@@ -131,12 +250,12 @@ int main(int argc, char* argv[]) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
// Testing correctness of round-trip (serialization + deserialization)
|
// Testing correctness of round-trip (serialization + deserialization)
|
||||||
std::string json_str = read_file(JSON_FILE);
|
simdjson::padded_string json_str = read_file(JSON_FILE);
|
||||||
|
|
||||||
// Loading up the data into a structure.
|
// Loading up the data into a structure.
|
||||||
simdjson::ondemand::parser parser;
|
simdjson::ondemand::parser parser;
|
||||||
simdjson::ondemand::document doc;
|
simdjson::ondemand::document doc;
|
||||||
if(parser.iterate(simdjson::pad(json_str)).get(doc)) {
|
if(parser.iterate(json_str).get(doc)) {
|
||||||
std::cerr << "Error loading the document!" << std::endl;
|
std::cerr << "Error loading the document!" << std::endl;
|
||||||
return EXIT_FAILURE;
|
return EXIT_FAILURE;
|
||||||
}
|
}
|
||||||
@@ -147,18 +266,37 @@ int main(int argc, char* argv[]) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// Benchmarking the serialization
|
// Benchmarking the serialization
|
||||||
|
// Note: simdjson benchmarks include both "fair" (fresh allocation) and "reuse" (buffer reuse) variants
|
||||||
|
// The "fair" variants allocate fresh memory each iteration, matching other libraries' behavior
|
||||||
|
// The "reuse" variants demonstrate the API's potential when buffer reuse is possible
|
||||||
|
|
||||||
if (matches_filter("nlohmann", filter)) {
|
if (matches_filter("nlohmann", filter)) {
|
||||||
bench_nlohmann(my_struct);
|
bench_nlohmann(my_struct);
|
||||||
}
|
}
|
||||||
|
#ifdef SIMDJSON_COMPETITION_YYJSON
|
||||||
|
if (matches_filter("yyjson", filter)) {
|
||||||
|
bench_yyjson(my_struct);
|
||||||
|
}
|
||||||
|
#endif
|
||||||
if (matches_filter("simdjson_static_reflection", filter)) {
|
if (matches_filter("simdjson_static_reflection", filter)) {
|
||||||
bench_simdjson_static_reflection(my_struct);
|
bench_simdjson_static_reflection(my_struct);
|
||||||
}
|
}
|
||||||
|
if (matches_filter("simdjson_reuse", filter)) {
|
||||||
|
bench_simdjson_static_reflection_reuse(my_struct);
|
||||||
|
}
|
||||||
|
#if SIMDJSON_STATIC_REFLECTION
|
||||||
|
if (matches_filter("simdjson_to", filter)) {
|
||||||
|
bench_simdjson_to(my_struct);
|
||||||
|
}
|
||||||
|
if (matches_filter("simdjson_to_reuse", filter)) {
|
||||||
|
bench_simdjson_to_reuse(my_struct);
|
||||||
|
}
|
||||||
|
#endif
|
||||||
#ifdef SIMDJSON_RUST_VERSION
|
#ifdef SIMDJSON_RUST_VERSION
|
||||||
if (matches_filter("rust", filter)) {
|
if (matches_filter("rust", filter)) {
|
||||||
printf("# WARNING: The Rust benchmark may not be directly comparable since it does not use an equivalent data structure.\n");
|
|
||||||
// Create a Rust-compatible CitmCatalog structure from the JSON string
|
// Create a Rust-compatible CitmCatalog structure from the JSON string
|
||||||
serde_benchmark::CitmCatalog* rust_data =
|
serde_benchmark::CitmCatalog* rust_data =
|
||||||
serde_benchmark::citm_from_str(json_str.c_str(), json_str.size());
|
serde_benchmark::citm_from_str(json_str.data(), json_str.size());
|
||||||
|
|
||||||
if (rust_data == nullptr) {
|
if (rust_data == nullptr) {
|
||||||
printf("# Failed to initialize Rust data structure\n");
|
printf("# Failed to initialize Rust data structure\n");
|
||||||
|
|||||||
@@ -4,79 +4,65 @@
|
|||||||
#include <string>
|
#include <string>
|
||||||
#include <vector>
|
#include <vector>
|
||||||
#include <map>
|
#include <map>
|
||||||
|
#include <optional>
|
||||||
|
#include <cstdint>
|
||||||
|
|
||||||
struct Area {
|
// Price structure - field names must match JSON keys for reflection
|
||||||
int64_t id;
|
struct CITMPrice {
|
||||||
std::string name;
|
uint64_t amount;
|
||||||
int64_t parent;
|
uint64_t audienceSubCategoryId;
|
||||||
std::vector<int64_t> childAreas;
|
uint64_t seatCategoryId;
|
||||||
bool operator==(const Area &other) const = default;
|
bool operator==(const CITMPrice&) const = default;
|
||||||
};
|
};
|
||||||
|
|
||||||
struct AudienceSubCategory {
|
struct CITMArea {
|
||||||
int64_t id;
|
uint64_t areaId;
|
||||||
std::string name;
|
std::vector<uint64_t> blockIds;
|
||||||
int64_t parent;
|
bool operator==(const CITMArea&) const = default;
|
||||||
bool operator==(const AudienceSubCategory &other) const = default;
|
|
||||||
};
|
};
|
||||||
|
|
||||||
struct Event {
|
struct CITMSeatCategory {
|
||||||
int64_t id;
|
std::vector<CITMArea> areas;
|
||||||
std::string name;
|
uint64_t seatCategoryId;
|
||||||
std::string description;
|
bool operator==(const CITMSeatCategory&) const = default;
|
||||||
int64_t subTopic;
|
|
||||||
int64_t topic;
|
|
||||||
std::vector<int64_t> audience;
|
|
||||||
bool operator==(const Event &other) const = default;
|
|
||||||
};
|
};
|
||||||
|
|
||||||
struct Performance {
|
struct CITMPerformance {
|
||||||
int64_t id;
|
uint64_t id;
|
||||||
std::string name;
|
uint64_t eventId;
|
||||||
int64_t event;
|
std::optional<std::string> logo;
|
||||||
std::string start;
|
std::optional<std::string> name;
|
||||||
int64_t venueCode;
|
std::vector<CITMPrice> prices;
|
||||||
bool operator==(const Performance &other) const = default;
|
std::vector<CITMSeatCategory> seatCategories;
|
||||||
|
std::optional<std::string> seatMapImage;
|
||||||
|
uint64_t start;
|
||||||
|
std::string venueCode;
|
||||||
|
bool operator==(const CITMPerformance&) const = default;
|
||||||
};
|
};
|
||||||
|
|
||||||
struct SeatCategory {
|
struct CITMEvent {
|
||||||
int64_t id;
|
uint64_t id;
|
||||||
std::string name;
|
std::string name;
|
||||||
std::vector<int64_t> areas;
|
std::optional<std::string> description;
|
||||||
bool operator==(const SeatCategory &other) const = default;
|
std::optional<std::string> logo;
|
||||||
};
|
std::vector<uint64_t> subTopicIds;
|
||||||
|
std::optional<std::string> subjectCode;
|
||||||
struct SubTopic {
|
std::optional<std::string> subtitle;
|
||||||
int64_t id;
|
std::vector<uint64_t> topicIds;
|
||||||
std::string name;
|
bool operator==(const CITMEvent&) const = default;
|
||||||
int64_t parent;
|
|
||||||
bool operator==(const SubTopic &other) const = default;
|
|
||||||
};
|
|
||||||
|
|
||||||
struct Topic {
|
|
||||||
int64_t id;
|
|
||||||
std::string name;
|
|
||||||
bool operator==(const Topic &other) const = default;
|
|
||||||
};
|
|
||||||
|
|
||||||
struct Venue {
|
|
||||||
int64_t id;
|
|
||||||
std::string name;
|
|
||||||
int64_t address;
|
|
||||||
bool operator==(const Venue &other) const = default;
|
|
||||||
};
|
};
|
||||||
|
|
||||||
struct CitmCatalog {
|
struct CitmCatalog {
|
||||||
std::map<std::string, Area> areas;
|
std::map<std::string, CITMEvent> events;
|
||||||
std::map<std::string, AudienceSubCategory> audienceSubCategory;
|
std::vector<CITMPerformance> performances;
|
||||||
std::map<std::string, Event> events;
|
bool operator==(const CitmCatalog&) const = default;
|
||||||
std::map<std::string, Performance> performances;
|
|
||||||
std::map<std::string, SeatCategory> seatCategory;
|
|
||||||
std::map<std::string, SubTopic> subTopic;
|
|
||||||
std::map<std::string, Topic> topic;
|
|
||||||
std::map<std::string, Venue> venue;
|
|
||||||
|
|
||||||
bool operator==(const CitmCatalog &other) const = default;
|
|
||||||
};
|
};
|
||||||
|
|
||||||
#endif
|
// Type aliases
|
||||||
|
using Event = CITMEvent;
|
||||||
|
using Performance = CITMPerformance;
|
||||||
|
using Price = CITMPrice;
|
||||||
|
using SeatArea = CITMArea;
|
||||||
|
using SeatCategoryInfo = CITMSeatCategory;
|
||||||
|
|
||||||
|
#endif
|
||||||
@@ -8,164 +8,72 @@
|
|||||||
|
|
||||||
using json = nlohmann::json;
|
using json = nlohmann::json;
|
||||||
|
|
||||||
// ---- Area ----
|
// ---- CITMPrice ----
|
||||||
inline void to_json(json &j, const Area &a) {
|
inline void to_json(json &j, const CITMPrice &p) {
|
||||||
j = json{
|
j = json{
|
||||||
{"id", a.id},
|
{"amount", p.amount},
|
||||||
{"name", a.name},
|
{"audienceSubCategoryId", p.audienceSubCategoryId},
|
||||||
{"parent", a.parent},
|
{"seatCategoryId", p.seatCategoryId}
|
||||||
{"childAreas", a.childAreas}
|
|
||||||
};
|
};
|
||||||
}
|
}
|
||||||
inline void from_json(const json &j, Area &a) {
|
|
||||||
j.at("id").get_to(a.id);
|
|
||||||
j.at("name").get_to(a.name);
|
|
||||||
j.at("parent").get_to(a.parent);
|
|
||||||
j.at("childAreas").get_to(a.childAreas);
|
|
||||||
}
|
|
||||||
|
|
||||||
// ---- AudienceSubCategory ----
|
// ---- CITMArea ----
|
||||||
inline void to_json(json &j, const AudienceSubCategory &asc) {
|
inline void to_json(json &j, const CITMArea &a) {
|
||||||
j = json{
|
j = json{
|
||||||
{"id", asc.id},
|
{"areaId", a.areaId},
|
||||||
{"name", asc.name},
|
{"blockIds", a.blockIds}
|
||||||
{"parent", asc.parent}
|
|
||||||
};
|
};
|
||||||
}
|
}
|
||||||
inline void from_json(const json &j, AudienceSubCategory &asc) {
|
|
||||||
j.at("id").get_to(asc.id);
|
|
||||||
j.at("name").get_to(asc.name);
|
|
||||||
j.at("parent").get_to(asc.parent);
|
|
||||||
}
|
|
||||||
|
|
||||||
// ---- Event ----
|
// ---- CITMSeatCategory ----
|
||||||
inline void to_json(json &j, const Event &e) {
|
inline void to_json(json &j, const CITMSeatCategory &s) {
|
||||||
j = json{
|
j = json{
|
||||||
{"id", e.id},
|
{"areas", s.areas},
|
||||||
{"name", e.name},
|
{"seatCategoryId", s.seatCategoryId}
|
||||||
{"description", e.description},
|
|
||||||
{"subTopic", e.subTopic},
|
|
||||||
{"topic", e.topic},
|
|
||||||
{"audience", e.audience}
|
|
||||||
};
|
};
|
||||||
}
|
}
|
||||||
inline void from_json(const json &j, Event &e) {
|
|
||||||
j.at("id").get_to(e.id);
|
|
||||||
j.at("name").get_to(e.name);
|
|
||||||
j.at("description").get_to(e.description);
|
|
||||||
j.at("subTopic").get_to(e.subTopic);
|
|
||||||
j.at("topic").get_to(e.topic);
|
|
||||||
j.at("audience").get_to(e.audience);
|
|
||||||
}
|
|
||||||
|
|
||||||
// ---- Performance ----
|
// ---- CITMPerformance ----
|
||||||
inline void to_json(json &j, const Performance &p) {
|
inline void to_json(json &j, const CITMPerformance &p) {
|
||||||
j = json{
|
j = json{
|
||||||
{"id", p.id},
|
{"id", p.id},
|
||||||
|
{"eventId", p.eventId},
|
||||||
|
{"logo", p.logo},
|
||||||
{"name", p.name},
|
{"name", p.name},
|
||||||
{"event", p.event},
|
{"prices", p.prices},
|
||||||
|
{"seatCategories", p.seatCategories},
|
||||||
|
{"seatMapImage", p.seatMapImage},
|
||||||
{"start", p.start},
|
{"start", p.start},
|
||||||
{"venueCode", p.venueCode}
|
{"venueCode", p.venueCode}
|
||||||
};
|
};
|
||||||
}
|
}
|
||||||
inline void from_json(const json &j, Performance &p) {
|
|
||||||
j.at("id").get_to(p.id);
|
|
||||||
j.at("name").get_to(p.name);
|
|
||||||
j.at("event").get_to(p.event);
|
|
||||||
j.at("start").get_to(p.start);
|
|
||||||
j.at("venueCode").get_to(p.venueCode);
|
|
||||||
}
|
|
||||||
|
|
||||||
// ---- SeatCategory ----
|
// ---- CITMEvent ----
|
||||||
inline void to_json(json &j, const SeatCategory &sc) {
|
inline void to_json(json &j, const CITMEvent &e) {
|
||||||
j = json{
|
j = json{
|
||||||
{"id", sc.id},
|
{"id", e.id},
|
||||||
{"name", sc.name},
|
{"name", e.name},
|
||||||
{"areas", sc.areas}
|
{"description", e.description},
|
||||||
|
{"logo", e.logo},
|
||||||
|
{"subTopicIds", e.subTopicIds},
|
||||||
|
{"subjectCode", e.subjectCode},
|
||||||
|
{"subtitle", e.subtitle},
|
||||||
|
{"topicIds", e.topicIds}
|
||||||
};
|
};
|
||||||
}
|
}
|
||||||
inline void from_json(const json &j, SeatCategory &sc) {
|
|
||||||
j.at("id").get_to(sc.id);
|
|
||||||
j.at("name").get_to(sc.name);
|
|
||||||
j.at("areas").get_to(sc.areas);
|
|
||||||
}
|
|
||||||
|
|
||||||
// ---- SubTopic ----
|
|
||||||
inline void to_json(json &j, const SubTopic &st) {
|
|
||||||
j = json{
|
|
||||||
{"id", st.id},
|
|
||||||
{"name", st.name},
|
|
||||||
{"parent", st.parent}
|
|
||||||
};
|
|
||||||
}
|
|
||||||
inline void from_json(const json &j, SubTopic &st) {
|
|
||||||
j.at("id").get_to(st.id);
|
|
||||||
j.at("name").get_to(st.name);
|
|
||||||
j.at("parent").get_to(st.parent);
|
|
||||||
}
|
|
||||||
|
|
||||||
// ---- Topic ----
|
|
||||||
inline void to_json(json &j, const Topic &t) {
|
|
||||||
j = json{
|
|
||||||
{"id", t.id},
|
|
||||||
{"name", t.name}
|
|
||||||
};
|
|
||||||
}
|
|
||||||
inline void from_json(const json &j, Topic &t) {
|
|
||||||
j.at("id").get_to(t.id);
|
|
||||||
j.at("name").get_to(t.name);
|
|
||||||
}
|
|
||||||
|
|
||||||
// ---- Venue ----
|
|
||||||
inline void to_json(json &j, const Venue &v) {
|
|
||||||
j = json{
|
|
||||||
{"id", v.id},
|
|
||||||
{"name", v.name},
|
|
||||||
{"address", v.address}
|
|
||||||
};
|
|
||||||
}
|
|
||||||
inline void from_json(const json &j, Venue &v) {
|
|
||||||
j.at("id").get_to(v.id);
|
|
||||||
j.at("name").get_to(v.name);
|
|
||||||
j.at("address").get_to(v.address);
|
|
||||||
}
|
|
||||||
|
|
||||||
// ---- CitmCatalog ----
|
// ---- CitmCatalog ----
|
||||||
inline void to_json(json &j, const CitmCatalog &c) {
|
inline void to_json(json &j, const CitmCatalog &c) {
|
||||||
j = json{
|
j = json{
|
||||||
{"areas", c.areas},
|
|
||||||
{"audienceSubCategory", c.audienceSubCategory},
|
|
||||||
{"events", c.events},
|
{"events", c.events},
|
||||||
{"performances", c.performances},
|
{"performances", c.performances}
|
||||||
{"seatCategory", c.seatCategory},
|
|
||||||
{"subTopic", c.subTopic},
|
|
||||||
{"topic", c.topic},
|
|
||||||
{"venue", c.venue}
|
|
||||||
};
|
};
|
||||||
}
|
}
|
||||||
inline void from_json(const json &j, CitmCatalog &c) {
|
|
||||||
j.at("areas").get_to(c.areas);
|
|
||||||
j.at("audienceSubCategory").get_to(c.audienceSubCategory);
|
|
||||||
j.at("events").get_to(c.events);
|
|
||||||
j.at("performances").get_to(c.performances);
|
|
||||||
j.at("seatCategory").get_to(c.seatCategory);
|
|
||||||
j.at("subTopic").get_to(c.subTopic);
|
|
||||||
j.at("topic").get_to(c.topic);
|
|
||||||
j.at("venue").get_to(c.venue);
|
|
||||||
}
|
|
||||||
|
|
||||||
// Optional convenience functions for benchmarking
|
// Serialization function
|
||||||
inline std::string nlohmann_serialize(const CitmCatalog &catalog) {
|
inline std::string nlohmann_serialize(const CitmCatalog &catalog) {
|
||||||
json j = catalog;
|
json j = catalog;
|
||||||
return j.dump();
|
return j.dump();
|
||||||
}
|
}
|
||||||
inline bool nlohmann_deserialize(const std::string &json_in, CitmCatalog &catalog) {
|
|
||||||
try {
|
|
||||||
catalog = json::parse(json_in);
|
|
||||||
return false; // success
|
|
||||||
} catch(...) {
|
|
||||||
return true; // failure
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
#endif // NLOHMANN_CITM_CATALOG_DATA_H
|
#endif // NLOHMANN_CITM_CATALOG_DATA_H
|
||||||
@@ -0,0 +1,189 @@
|
|||||||
|
#ifndef RAPIDJSON_CITM_CATALOG_DATA_H
|
||||||
|
#define RAPIDJSON_CITM_CATALOG_DATA_H
|
||||||
|
|
||||||
|
#include "citm_catalog_data.h"
|
||||||
|
#include <rapidjson/document.h>
|
||||||
|
#include <rapidjson/writer.h>
|
||||||
|
#include <rapidjson/stringbuffer.h>
|
||||||
|
#include <rapidjson/error/en.h>
|
||||||
|
|
||||||
|
using namespace rapidjson;
|
||||||
|
|
||||||
|
// RapidJSON deserialization for CITM Catalog data
|
||||||
|
CitmCatalog rapidjson_deserialize_citm(const std::string& json_str) {
|
||||||
|
Document doc;
|
||||||
|
doc.Parse(json_str.c_str());
|
||||||
|
|
||||||
|
if (doc.HasParseError()) {
|
||||||
|
throw std::runtime_error("RapidJSON parse error");
|
||||||
|
}
|
||||||
|
|
||||||
|
CitmCatalog catalog;
|
||||||
|
|
||||||
|
// Parse events
|
||||||
|
if (doc.HasMember("events") && doc["events"].IsObject()) {
|
||||||
|
const Value& events = doc["events"];
|
||||||
|
for (auto it = events.MemberBegin(); it != events.MemberEnd(); ++it) {
|
||||||
|
Event event;
|
||||||
|
const Value& ev = it->value;
|
||||||
|
|
||||||
|
if (ev.HasMember("description") && ev["description"].IsString())
|
||||||
|
event.description = ev["description"].GetString();
|
||||||
|
if (ev.HasMember("id") && ev["id"].IsUint64())
|
||||||
|
event.id = ev["id"].GetUint64();
|
||||||
|
if (ev.HasMember("logo") && ev["logo"].IsString())
|
||||||
|
event.logo = ev["logo"].GetString();
|
||||||
|
if (ev.HasMember("name") && ev["name"].IsString())
|
||||||
|
event.name = ev["name"].GetString();
|
||||||
|
if (ev.HasMember("subjectCode") && ev["subjectCode"].IsString())
|
||||||
|
event.subjectCode = ev["subjectCode"].GetString();
|
||||||
|
if (ev.HasMember("subtitle") && ev["subtitle"].IsString())
|
||||||
|
event.subtitle = ev["subtitle"].GetString();
|
||||||
|
|
||||||
|
if (ev.HasMember("topicIds") && ev["topicIds"].IsArray()) {
|
||||||
|
const Value& topics = ev["topicIds"];
|
||||||
|
for (SizeType j = 0; j < topics.Size(); j++) {
|
||||||
|
if (topics[j].IsUint64())
|
||||||
|
event.topicIds.push_back(topics[j].GetUint64());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
if (ev.HasMember("subTopicIds") && ev["subTopicIds"].IsArray()) {
|
||||||
|
const Value& subtopics = ev["subTopicIds"];
|
||||||
|
for (SizeType j = 0; j < subtopics.Size(); j++) {
|
||||||
|
if (subtopics[j].IsUint64())
|
||||||
|
event.subTopicIds.push_back(subtopics[j].GetUint64());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
catalog.events[it->name.GetString()] = event;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Parse performances
|
||||||
|
if (doc.HasMember("performances") && doc["performances"].IsArray()) {
|
||||||
|
const Value& performances = doc["performances"];
|
||||||
|
for (SizeType i = 0; i < performances.Size(); i++) {
|
||||||
|
Performance perf;
|
||||||
|
const Value& p = performances[i];
|
||||||
|
|
||||||
|
if (p.HasMember("id") && p["id"].IsUint64())
|
||||||
|
perf.id = p["id"].GetUint64();
|
||||||
|
if (p.HasMember("eventId") && p["eventId"].IsUint64())
|
||||||
|
perf.eventId = p["eventId"].GetUint64();
|
||||||
|
if (p.HasMember("start") && p["start"].IsUint64())
|
||||||
|
perf.start = p["start"].GetUint64();
|
||||||
|
if (p.HasMember("venueCode") && p["venueCode"].IsString())
|
||||||
|
perf.venueCode = p["venueCode"].GetString();
|
||||||
|
if (p.HasMember("name") && p["name"].IsString())
|
||||||
|
perf.name = p["name"].GetString();
|
||||||
|
|
||||||
|
catalog.performances.push_back(perf);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Parse other string maps
|
||||||
|
auto parseStringMap = [&doc](const char* key, std::map<std::string, std::string>& target) {
|
||||||
|
if (doc.HasMember(key) && doc[key].IsObject()) {
|
||||||
|
const Value& obj = doc[key];
|
||||||
|
for (auto it = obj.MemberBegin(); it != obj.MemberEnd(); ++it) {
|
||||||
|
if (it->value.IsString()) {
|
||||||
|
target[it->name.GetString()] = it->value.GetString();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
return catalog;
|
||||||
|
}
|
||||||
|
|
||||||
|
// RapidJSON serialization for CITM Catalog data
|
||||||
|
std::string rapidjson_serialize_citm(const CitmCatalog& catalog) {
|
||||||
|
Document doc;
|
||||||
|
doc.SetObject();
|
||||||
|
Document::AllocatorType& allocator = doc.GetAllocator();
|
||||||
|
|
||||||
|
// Serialize events
|
||||||
|
Value events_obj(kObjectType);
|
||||||
|
for (const auto& [key, event] : catalog.events) {
|
||||||
|
Value event_obj(kObjectType);
|
||||||
|
|
||||||
|
if (event.description) {
|
||||||
|
Value desc;
|
||||||
|
desc.SetString(event.description->c_str(), allocator);
|
||||||
|
event_obj.AddMember("description", desc, allocator);
|
||||||
|
}
|
||||||
|
|
||||||
|
event_obj.AddMember("id", event.id, allocator);
|
||||||
|
|
||||||
|
if (event.logo) {
|
||||||
|
Value logo;
|
||||||
|
logo.SetString(event.logo->c_str(), allocator);
|
||||||
|
event_obj.AddMember("logo", logo, allocator);
|
||||||
|
}
|
||||||
|
|
||||||
|
Value name;
|
||||||
|
name.SetString(event.name.c_str(), allocator);
|
||||||
|
event_obj.AddMember("name", name, allocator);
|
||||||
|
|
||||||
|
if (event.subjectCode) {
|
||||||
|
Value subject;
|
||||||
|
subject.SetString(event.subjectCode->c_str(), allocator);
|
||||||
|
event_obj.AddMember("subjectCode", subject, allocator);
|
||||||
|
}
|
||||||
|
|
||||||
|
if (event.subtitle) {
|
||||||
|
Value subtitle;
|
||||||
|
subtitle.SetString(event.subtitle->c_str(), allocator);
|
||||||
|
event_obj.AddMember("subtitle", subtitle, allocator);
|
||||||
|
}
|
||||||
|
|
||||||
|
Value topicIds(kArrayType);
|
||||||
|
for (uint64_t id : event.topicIds) {
|
||||||
|
topicIds.PushBack(id, allocator);
|
||||||
|
}
|
||||||
|
event_obj.AddMember("topicIds", topicIds, allocator);
|
||||||
|
|
||||||
|
Value subTopicIds(kArrayType);
|
||||||
|
for (uint64_t id : event.subTopicIds) {
|
||||||
|
subTopicIds.PushBack(id, allocator);
|
||||||
|
}
|
||||||
|
event_obj.AddMember("subTopicIds", subTopicIds, allocator);
|
||||||
|
|
||||||
|
Value key_val;
|
||||||
|
key_val.SetString(key.c_str(), allocator);
|
||||||
|
events_obj.AddMember(key_val, event_obj, allocator);
|
||||||
|
}
|
||||||
|
doc.AddMember("events", events_obj, allocator);
|
||||||
|
|
||||||
|
// Serialize performances
|
||||||
|
Value performances_array(kArrayType);
|
||||||
|
for (const auto& perf : catalog.performances) {
|
||||||
|
Value perf_obj(kObjectType);
|
||||||
|
perf_obj.AddMember("id", perf.id, allocator);
|
||||||
|
perf_obj.AddMember("eventId", perf.eventId, allocator);
|
||||||
|
perf_obj.AddMember("start", perf.start, allocator);
|
||||||
|
|
||||||
|
Value venue;
|
||||||
|
venue.SetString(perf.venueCode.c_str(), allocator);
|
||||||
|
perf_obj.AddMember("venueCode", venue, allocator);
|
||||||
|
|
||||||
|
if (perf.name) {
|
||||||
|
Value name;
|
||||||
|
name.SetString(perf.name->c_str(), allocator);
|
||||||
|
perf_obj.AddMember("name", name, allocator);
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
performances_array.PushBack(perf_obj, allocator);
|
||||||
|
}
|
||||||
|
doc.AddMember("performances", performances_array, allocator);
|
||||||
|
|
||||||
|
StringBuffer buffer;
|
||||||
|
Writer<StringBuffer> writer(buffer);
|
||||||
|
doc.Accept(writer);
|
||||||
|
|
||||||
|
return buffer.GetString();
|
||||||
|
}
|
||||||
|
|
||||||
|
#endif // RAPIDJSON_CITM_CATALOG_DATA_H
|
||||||
@@ -0,0 +1,231 @@
|
|||||||
|
#ifndef YYJSON_CITM_CATALOG_DATA_H
|
||||||
|
#define YYJSON_CITM_CATALOG_DATA_H
|
||||||
|
|
||||||
|
#include "citm_catalog_data.h"
|
||||||
|
#include <yyjson.h>
|
||||||
|
#include <string>
|
||||||
|
#include <stdexcept>
|
||||||
|
|
||||||
|
// yyjson deserialization for CITM Catalog data
|
||||||
|
// Matches C++ CitmCatalog struct (only events + performances)
|
||||||
|
CitmCatalog yyjson_deserialize_citm(const std::string &json_str) {
|
||||||
|
CitmCatalog catalog;
|
||||||
|
|
||||||
|
yyjson_doc *doc = yyjson_read(json_str.c_str(), json_str.size(), 0);
|
||||||
|
if (!doc) {
|
||||||
|
throw std::runtime_error("yyjson parse error");
|
||||||
|
}
|
||||||
|
|
||||||
|
yyjson_val *root = yyjson_doc_get_root(doc);
|
||||||
|
if (!root) {
|
||||||
|
yyjson_doc_free(doc);
|
||||||
|
return catalog;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Parse events
|
||||||
|
yyjson_val *events_val = yyjson_obj_get(root, "events");
|
||||||
|
if (events_val && yyjson_is_obj(events_val)) {
|
||||||
|
size_t idx, max;
|
||||||
|
yyjson_val *key, *val;
|
||||||
|
yyjson_obj_foreach(events_val, idx, max, key, val) {
|
||||||
|
CITMEvent event;
|
||||||
|
|
||||||
|
yyjson_val *v;
|
||||||
|
v = yyjson_obj_get(val, "description");
|
||||||
|
if (v && yyjson_is_str(v)) event.description = yyjson_get_str(v);
|
||||||
|
|
||||||
|
v = yyjson_obj_get(val, "id");
|
||||||
|
if (v && yyjson_is_uint(v)) event.id = yyjson_get_uint(v);
|
||||||
|
|
||||||
|
v = yyjson_obj_get(val, "logo");
|
||||||
|
if (v && yyjson_is_str(v)) event.logo = yyjson_get_str(v);
|
||||||
|
|
||||||
|
v = yyjson_obj_get(val, "name");
|
||||||
|
if (v && yyjson_is_str(v)) event.name = yyjson_get_str(v);
|
||||||
|
|
||||||
|
v = yyjson_obj_get(val, "subjectCode");
|
||||||
|
if (v && yyjson_is_str(v)) event.subjectCode = yyjson_get_str(v);
|
||||||
|
|
||||||
|
v = yyjson_obj_get(val, "subtitle");
|
||||||
|
if (v && yyjson_is_str(v)) event.subtitle = yyjson_get_str(v);
|
||||||
|
|
||||||
|
// Parse topicIds array
|
||||||
|
v = yyjson_obj_get(val, "topicIds");
|
||||||
|
if (v && yyjson_is_arr(v)) {
|
||||||
|
size_t arr_idx, arr_max;
|
||||||
|
yyjson_val *arr_val;
|
||||||
|
yyjson_arr_foreach(v, arr_idx, arr_max, arr_val) {
|
||||||
|
if (yyjson_is_uint(arr_val))
|
||||||
|
event.topicIds.push_back(yyjson_get_uint(arr_val));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Parse subTopicIds array
|
||||||
|
v = yyjson_obj_get(val, "subTopicIds");
|
||||||
|
if (v && yyjson_is_arr(v)) {
|
||||||
|
size_t arr_idx, arr_max;
|
||||||
|
yyjson_val *arr_val;
|
||||||
|
yyjson_arr_foreach(v, arr_idx, arr_max, arr_val) {
|
||||||
|
if (yyjson_is_uint(arr_val))
|
||||||
|
event.subTopicIds.push_back(yyjson_get_uint(arr_val));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
if (yyjson_is_str(key))
|
||||||
|
catalog.events[yyjson_get_str(key)] = event;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Parse performances (simplified - full parsing would need prices/seatCategories)
|
||||||
|
yyjson_val *performances_val = yyjson_obj_get(root, "performances");
|
||||||
|
if (performances_val && yyjson_is_arr(performances_val)) {
|
||||||
|
size_t idx, max;
|
||||||
|
yyjson_val *perf_val;
|
||||||
|
yyjson_arr_foreach(performances_val, idx, max, perf_val) {
|
||||||
|
CITMPerformance perf;
|
||||||
|
|
||||||
|
yyjson_val *v;
|
||||||
|
v = yyjson_obj_get(perf_val, "id");
|
||||||
|
if (v && yyjson_is_uint(v)) perf.id = yyjson_get_uint(v);
|
||||||
|
|
||||||
|
v = yyjson_obj_get(perf_val, "eventId");
|
||||||
|
if (v && yyjson_is_uint(v)) perf.eventId = yyjson_get_uint(v);
|
||||||
|
|
||||||
|
v = yyjson_obj_get(perf_val, "start");
|
||||||
|
if (v && yyjson_is_uint(v)) perf.start = yyjson_get_uint(v);
|
||||||
|
|
||||||
|
v = yyjson_obj_get(perf_val, "venueCode");
|
||||||
|
if (v && yyjson_is_str(v)) perf.venueCode = yyjson_get_str(v);
|
||||||
|
|
||||||
|
v = yyjson_obj_get(perf_val, "name");
|
||||||
|
if (v && yyjson_is_str(v)) perf.name = yyjson_get_str(v);
|
||||||
|
|
||||||
|
v = yyjson_obj_get(perf_val, "logo");
|
||||||
|
if (v && yyjson_is_str(v)) perf.logo = yyjson_get_str(v);
|
||||||
|
|
||||||
|
v = yyjson_obj_get(perf_val, "seatMapImage");
|
||||||
|
if (v && yyjson_is_str(v)) perf.seatMapImage = yyjson_get_str(v);
|
||||||
|
|
||||||
|
// Note: prices and seatCategories parsing omitted for brevity
|
||||||
|
// The serialization benchmark uses data loaded by simdjson
|
||||||
|
|
||||||
|
catalog.performances.push_back(perf);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
yyjson_doc_free(doc);
|
||||||
|
return catalog;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Helper to add optional string field
|
||||||
|
static inline void yyjson_add_optional_str(yyjson_mut_doc *doc, yyjson_mut_val *obj,
|
||||||
|
const char *key, const std::optional<std::string> &val) {
|
||||||
|
if (val.has_value()) {
|
||||||
|
yyjson_mut_obj_add_str(doc, obj, key, val->c_str());
|
||||||
|
} else {
|
||||||
|
yyjson_mut_obj_add_null(doc, obj, key);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// yyjson serialization for CITM Catalog data
|
||||||
|
// Matches C++ CitmCatalog struct exactly (only events + performances)
|
||||||
|
std::string yyjson_serialize_citm(const CitmCatalog &catalog) {
|
||||||
|
yyjson_mut_doc *doc = yyjson_mut_doc_new(NULL);
|
||||||
|
yyjson_mut_val *root = yyjson_mut_obj(doc);
|
||||||
|
yyjson_mut_doc_set_root(doc, root);
|
||||||
|
|
||||||
|
// Create events object
|
||||||
|
yyjson_mut_val *events_obj = yyjson_mut_obj(doc);
|
||||||
|
for (const auto& [key, event] : catalog.events) {
|
||||||
|
yyjson_mut_val *event_obj = yyjson_mut_obj(doc);
|
||||||
|
|
||||||
|
yyjson_add_optional_str(doc, event_obj, "description", event.description);
|
||||||
|
yyjson_mut_obj_add_uint(doc, event_obj, "id", event.id);
|
||||||
|
yyjson_add_optional_str(doc, event_obj, "logo", event.logo);
|
||||||
|
// name is not optional in CITMEvent
|
||||||
|
yyjson_mut_obj_add_str(doc, event_obj, "name", event.name.c_str());
|
||||||
|
|
||||||
|
// Add subTopicIds array
|
||||||
|
yyjson_mut_val *subtopic_ids = yyjson_mut_arr(doc);
|
||||||
|
for (uint64_t id : event.subTopicIds) {
|
||||||
|
yyjson_mut_arr_add_uint(doc, subtopic_ids, id);
|
||||||
|
}
|
||||||
|
yyjson_mut_obj_add_val(doc, event_obj, "subTopicIds", subtopic_ids);
|
||||||
|
|
||||||
|
yyjson_add_optional_str(doc, event_obj, "subjectCode", event.subjectCode);
|
||||||
|
yyjson_add_optional_str(doc, event_obj, "subtitle", event.subtitle);
|
||||||
|
|
||||||
|
// Add topicIds array
|
||||||
|
yyjson_mut_val *topic_ids = yyjson_mut_arr(doc);
|
||||||
|
for (uint64_t id : event.topicIds) {
|
||||||
|
yyjson_mut_arr_add_uint(doc, topic_ids, id);
|
||||||
|
}
|
||||||
|
yyjson_mut_obj_add_val(doc, event_obj, "topicIds", topic_ids);
|
||||||
|
|
||||||
|
yyjson_mut_obj_add_val(doc, events_obj, key.c_str(), event_obj);
|
||||||
|
}
|
||||||
|
yyjson_mut_obj_add_val(doc, root, "events", events_obj);
|
||||||
|
|
||||||
|
// Create performances array
|
||||||
|
yyjson_mut_val *performances_array = yyjson_mut_arr(doc);
|
||||||
|
for (const auto& perf : catalog.performances) {
|
||||||
|
yyjson_mut_val *perf_obj = yyjson_mut_obj(doc);
|
||||||
|
|
||||||
|
yyjson_mut_obj_add_uint(doc, perf_obj, "eventId", perf.eventId);
|
||||||
|
yyjson_mut_obj_add_uint(doc, perf_obj, "id", perf.id);
|
||||||
|
yyjson_add_optional_str(doc, perf_obj, "logo", perf.logo);
|
||||||
|
yyjson_add_optional_str(doc, perf_obj, "name", perf.name);
|
||||||
|
|
||||||
|
// Add prices array
|
||||||
|
yyjson_mut_val *prices_array = yyjson_mut_arr(doc);
|
||||||
|
for (const auto& price : perf.prices) {
|
||||||
|
yyjson_mut_val *price_obj = yyjson_mut_obj(doc);
|
||||||
|
yyjson_mut_obj_add_uint(doc, price_obj, "amount", price.amount);
|
||||||
|
yyjson_mut_obj_add_uint(doc, price_obj, "audienceSubCategoryId", price.audienceSubCategoryId);
|
||||||
|
yyjson_mut_obj_add_uint(doc, price_obj, "seatCategoryId", price.seatCategoryId);
|
||||||
|
yyjson_mut_arr_append(prices_array, price_obj);
|
||||||
|
}
|
||||||
|
yyjson_mut_obj_add_val(doc, perf_obj, "prices", prices_array);
|
||||||
|
|
||||||
|
// Add seatCategories array
|
||||||
|
yyjson_mut_val *seat_cats_array = yyjson_mut_arr(doc);
|
||||||
|
for (const auto& seatCat : perf.seatCategories) {
|
||||||
|
yyjson_mut_val *seat_cat_obj = yyjson_mut_obj(doc);
|
||||||
|
|
||||||
|
// Add areas array
|
||||||
|
yyjson_mut_val *areas_array = yyjson_mut_arr(doc);
|
||||||
|
for (const auto& area : seatCat.areas) {
|
||||||
|
yyjson_mut_val *area_obj = yyjson_mut_obj(doc);
|
||||||
|
yyjson_mut_obj_add_uint(doc, area_obj, "areaId", area.areaId);
|
||||||
|
|
||||||
|
yyjson_mut_val *block_ids = yyjson_mut_arr(doc);
|
||||||
|
for (uint64_t blockId : area.blockIds) {
|
||||||
|
yyjson_mut_arr_add_uint(doc, block_ids, blockId);
|
||||||
|
}
|
||||||
|
yyjson_mut_obj_add_val(doc, area_obj, "blockIds", block_ids);
|
||||||
|
yyjson_mut_arr_append(areas_array, area_obj);
|
||||||
|
}
|
||||||
|
yyjson_mut_obj_add_val(doc, seat_cat_obj, "areas", areas_array);
|
||||||
|
yyjson_mut_obj_add_uint(doc, seat_cat_obj, "seatCategoryId", seatCat.seatCategoryId);
|
||||||
|
yyjson_mut_arr_append(seat_cats_array, seat_cat_obj);
|
||||||
|
}
|
||||||
|
yyjson_mut_obj_add_val(doc, perf_obj, "seatCategories", seat_cats_array);
|
||||||
|
|
||||||
|
yyjson_add_optional_str(doc, perf_obj, "seatMapImage", perf.seatMapImage);
|
||||||
|
yyjson_mut_obj_add_uint(doc, perf_obj, "start", perf.start);
|
||||||
|
yyjson_mut_obj_add_str(doc, perf_obj, "venueCode", perf.venueCode.c_str());
|
||||||
|
|
||||||
|
yyjson_mut_arr_append(performances_array, perf_obj);
|
||||||
|
}
|
||||||
|
yyjson_mut_obj_add_val(doc, root, "performances", performances_array);
|
||||||
|
|
||||||
|
// Write to string
|
||||||
|
char *json_output = yyjson_mut_write(doc, 0, NULL);
|
||||||
|
std::string result(json_output);
|
||||||
|
free(json_output);
|
||||||
|
yyjson_mut_doc_free(doc);
|
||||||
|
|
||||||
|
return result;
|
||||||
|
}
|
||||||
|
|
||||||
|
#endif // YYJSON_CITM_CATALOG_DATA_H
|
||||||
@@ -5,405 +5,165 @@ extern crate libc;
|
|||||||
use libc::{c_char, size_t};
|
use libc::{c_char, size_t};
|
||||||
use serde::{Serialize, Deserialize};
|
use serde::{Serialize, Deserialize};
|
||||||
use std::{collections::HashMap, ffi::CString, ptr, slice};
|
use std::{collections::HashMap, ffi::CString, ptr, slice};
|
||||||
use serde::de::{self, Deserializer};
|
|
||||||
/******************************************************/
|
|
||||||
/******************************************************/
|
|
||||||
/**
|
|
||||||
* Warning: the C++ code may not generate the same JSON.
|
|
||||||
*/
|
|
||||||
/******************************************************/
|
|
||||||
/******************************************************/
|
|
||||||
|
|
||||||
// This has no equivalent in C++:
|
//==============================================================================
|
||||||
#[derive(Serialize, Deserialize)]
|
// Twitter Benchmark Structures
|
||||||
pub struct Metadata {
|
// These match the C++ TwitterData structures exactly
|
||||||
result_type: String,
|
//==============================================================================
|
||||||
iso_language_code: String,
|
|
||||||
}
|
|
||||||
|
|
||||||
#[derive(Serialize, Deserialize)]
|
#[derive(Serialize, Deserialize)]
|
||||||
pub struct User {
|
pub struct User {
|
||||||
id: i64,
|
id: u64,
|
||||||
id_str: String,
|
|
||||||
name: String,
|
name: String,
|
||||||
screen_name: String,
|
screen_name: String,
|
||||||
location: String,
|
location: String,
|
||||||
description: String,
|
description: String,
|
||||||
// C++ does not have those:
|
|
||||||
// url: Option<String>,
|
|
||||||
//protected: bool,
|
|
||||||
//listed_count: i64,
|
|
||||||
//created_at: String,
|
|
||||||
//favourites_count: i64,
|
|
||||||
//utc_offset: Option<i64>,
|
|
||||||
//time_zone: Option<String>,
|
|
||||||
//geo_enabled: bool,
|
|
||||||
verified: bool,
|
verified: bool,
|
||||||
followers_count: i64,
|
followers_count: u64,
|
||||||
friends_count: i64,
|
friends_count: u64,
|
||||||
statuses_count: i64,
|
statuses_count: u64,
|
||||||
// C++ does not have those:
|
|
||||||
//lang: String,
|
|
||||||
//profile_background_color: String,
|
|
||||||
//profile_background_image_url: String,
|
|
||||||
//profile_background_image_url_https: String,
|
|
||||||
//profile_background_tile: bool,
|
|
||||||
//profile_image_url: String,
|
|
||||||
//profile_image_url_https: String,
|
|
||||||
//profile_banner_url: Option<String>,
|
|
||||||
//profile_link_color: String,
|
|
||||||
//profile_sidebar_border_color: String,
|
|
||||||
//profile_sidebar_fill_color: String,
|
|
||||||
//profile_text_color: String,
|
|
||||||
//profile_use_background_image: bool,
|
|
||||||
//default_profile: bool,
|
|
||||||
//default_profile_image: bool,
|
|
||||||
//following: bool,
|
|
||||||
//follow_request_sent: bool,
|
|
||||||
//notifications: bool,
|
|
||||||
}
|
|
||||||
|
|
||||||
#[derive(Serialize, Deserialize)]
|
|
||||||
pub struct Hashtag {
|
|
||||||
text: String,
|
|
||||||
|
|
||||||
// C++ has those but D. Lemire does not know what they are, they don't appear in the JSON:
|
|
||||||
// int64_t indices_start;
|
|
||||||
// int64_t indices_end;
|
|
||||||
}
|
|
||||||
|
|
||||||
#[derive(Serialize, Deserialize)]
|
|
||||||
pub struct Url {
|
|
||||||
url: String,
|
|
||||||
expanded_url: String,
|
|
||||||
display_url: String,
|
|
||||||
// C++ has those but D. Lemire does not know what they are, they don't appear in the JSON:
|
|
||||||
// int64_t indices_start;
|
|
||||||
// int64_t indices_end;
|
|
||||||
}
|
|
||||||
|
|
||||||
#[derive(Serialize, Deserialize)]
|
|
||||||
pub struct UserMention {
|
|
||||||
id: i64,
|
|
||||||
name: String,
|
|
||||||
screen_name: String,
|
|
||||||
// Not in the C++ equivalent:
|
|
||||||
//id_str: String,
|
|
||||||
//indices: Vec<i64>,
|
|
||||||
// C++ has those but D. Lemire does not know what they are, they don't appear in the JSON:
|
|
||||||
// int64_t indices_start;
|
|
||||||
// int64_t indices_end;
|
|
||||||
}
|
|
||||||
|
|
||||||
#[derive(Serialize, Deserialize)]
|
|
||||||
pub struct Entities {
|
|
||||||
hashtags: Vec<Hashtag>,
|
|
||||||
urls: Vec<Url>,
|
|
||||||
user_mentions: Vec<UserMention>,
|
|
||||||
}
|
}
|
||||||
|
|
||||||
#[derive(Serialize, Deserialize)]
|
#[derive(Serialize, Deserialize)]
|
||||||
pub struct Status {
|
pub struct Status {
|
||||||
created_at: String,
|
created_at: String,
|
||||||
id: i64,
|
id: u64,
|
||||||
text: String,
|
text: String,
|
||||||
user: User,
|
user: User,
|
||||||
entities: Entities,
|
retweet_count: u64,
|
||||||
retweet_count: i64,
|
favorite_count: u64,
|
||||||
favorite_count: i64,
|
|
||||||
favorited: bool,
|
|
||||||
retweeted: bool,
|
|
||||||
// None of these are in the C++ equivalent:
|
|
||||||
/*
|
|
||||||
metadata: Metadata,
|
|
||||||
id_str: String,
|
|
||||||
source: String,
|
|
||||||
truncated: bool,
|
|
||||||
in_reply_to_status_id: Option<i64>,
|
|
||||||
in_reply_to_status_id_str: Option<String>,
|
|
||||||
in_reply_to_user_id: Option<i64>,
|
|
||||||
in_reply_to_user_id_str: Option<String>,
|
|
||||||
in_reply_to_screen_name: Option<String>,
|
|
||||||
geo: Option<String>,
|
|
||||||
coordinates: Option<String>,
|
|
||||||
place: Option<String>,
|
|
||||||
contributors: Option<String>,
|
|
||||||
lang: String,
|
|
||||||
*/
|
|
||||||
}
|
}
|
||||||
|
|
||||||
#[derive(Serialize, Deserialize)]
|
#[derive(Serialize, Deserialize)]
|
||||||
pub struct TwitterData {
|
pub struct TwitterData {
|
||||||
statuses: Vec<Status>,
|
statuses: Vec<Status>,
|
||||||
}
|
}
|
||||||
|
static mut TWITTER_DATA: *mut TwitterData = std::ptr::null_mut();
|
||||||
|
|
||||||
#[no_mangle]
|
#[no_mangle]
|
||||||
pub unsafe extern "C" fn twitter_from_str(raw_input: *const c_char, raw_input_length: size_t) -> *mut TwitterData {
|
pub unsafe extern "C" fn twitter_from_str(raw_input: *const c_char, raw_input_length: size_t) -> *mut TwitterData {
|
||||||
let input = std::str::from_utf8_unchecked(slice::from_raw_parts(raw_input as *const u8, raw_input_length));
|
let input = std::str::from_utf8_unchecked(slice::from_raw_parts(raw_input as *const u8, raw_input_length));
|
||||||
match serde_json::from_str(&input) {
|
match serde_json::from_str(&input) {
|
||||||
Ok(result) => Box::into_raw(Box::new(result)),
|
Ok(result) => Box::into_raw(Box::new(result)),
|
||||||
Err(_) => std::ptr::null_mut(),
|
Err(_) => std::ptr::null_mut(),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
#[no_mangle]
|
#[no_mangle]
|
||||||
pub unsafe extern "C" fn str_from_twitter(raw: *mut TwitterData) -> *const c_char {
|
pub unsafe extern "C" fn set_twitter_data(raw: *mut TwitterData) {
|
||||||
let twitter_thing = { &*raw };
|
TWITTER_DATA = raw;
|
||||||
let serialized = serde_json::to_string(&twitter_thing).unwrap();
|
|
||||||
return std::ffi::CString::new(serialized.as_str()).unwrap().into_raw()
|
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[no_mangle]
|
||||||
|
pub unsafe extern "C" fn serialize_twitter_to_string() -> usize {
|
||||||
|
if TWITTER_DATA.is_null() {
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
let data = &*TWITTER_DATA;
|
||||||
|
serde_json::to_string(data).unwrap().len()
|
||||||
|
}
|
||||||
|
|
||||||
#[no_mangle]
|
#[no_mangle]
|
||||||
pub unsafe extern "C" fn free_twitter(raw: *mut TwitterData) {
|
pub unsafe extern "C" fn free_twitter(raw: *mut TwitterData) {
|
||||||
if raw.is_null() {
|
if raw.is_null() {
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
|
drop(Box::from_raw(raw))
|
||||||
drop(Box::from_raw(raw))
|
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
#[no_mangle]
|
#[no_mangle]
|
||||||
pub unsafe extern fn free_string(ptr: *const c_char) {
|
pub unsafe extern fn free_string(ptr: *const c_char) {
|
||||||
let _ = std::ffi::CString::from_raw(ptr as *mut _);
|
let _ = std::ffi::CString::from_raw(ptr as *mut _);
|
||||||
}
|
}
|
||||||
|
|
||||||
// Functions associated with the CitmCatalog benchmark
|
//==============================================================================
|
||||||
|
// CITM Catalog Benchmark Structures
|
||||||
|
// These match the C++ CitmCatalog structures EXACTLY for fair comparison
|
||||||
|
//==============================================================================
|
||||||
|
|
||||||
#[derive(Serialize, Deserialize)]
|
/// Matches C++ CITMPrice struct exactly
|
||||||
pub struct Area {
|
#[derive(Serialize, Deserialize, Debug, Clone)]
|
||||||
pub id: i64,
|
pub struct CITMPrice {
|
||||||
pub name: Option<String>, // Changed to Option
|
pub amount: u64,
|
||||||
pub parent: i64,
|
#[serde(rename = "audienceSubCategoryId")]
|
||||||
#[serde(rename = "childAreas")]
|
pub audience_sub_category_id: u64,
|
||||||
pub child_areas: Vec<i64>,
|
#[serde(rename = "seatCategoryId")]
|
||||||
|
pub seat_category_id: u64,
|
||||||
}
|
}
|
||||||
|
|
||||||
#[derive(Serialize, Deserialize)]
|
/// Matches C++ CITMArea struct exactly
|
||||||
pub struct AudienceSubCategory {
|
#[derive(Serialize, Deserialize, Debug, Clone)]
|
||||||
pub id: i64,
|
pub struct CITMArea {
|
||||||
pub name: Option<String>, // Changed to Option
|
#[serde(rename = "areaId")]
|
||||||
pub parent: i64,
|
pub area_id: u64,
|
||||||
|
#[serde(rename = "blockIds")]
|
||||||
|
pub block_ids: Vec<u64>,
|
||||||
}
|
}
|
||||||
|
|
||||||
#[derive(Serialize, Deserialize, Debug)]
|
/// Matches C++ CITMSeatCategory struct exactly
|
||||||
pub struct Event {
|
#[derive(Serialize, Deserialize, Debug, Clone)]
|
||||||
#[serde(default)]
|
pub struct CITMSeatCategory {
|
||||||
pub description: Option<String>,
|
pub areas: Vec<CITMArea>,
|
||||||
pub id: i64,
|
#[serde(rename = "seatCategoryId")]
|
||||||
|
pub seat_category_id: u64,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Matches C++ CITMPerformance struct exactly
|
||||||
|
#[derive(Serialize, Deserialize, Debug, Clone)]
|
||||||
|
pub struct CITMPerformance {
|
||||||
|
pub id: u64,
|
||||||
|
#[serde(rename = "eventId")]
|
||||||
|
pub event_id: u64,
|
||||||
#[serde(default)]
|
#[serde(default)]
|
||||||
pub logo: Option<String>,
|
pub logo: Option<String>,
|
||||||
#[serde(default)]
|
#[serde(default)]
|
||||||
pub name: Option<String>,
|
pub name: Option<String>,
|
||||||
|
pub prices: Vec<CITMPrice>,
|
||||||
|
#[serde(rename = "seatCategories")]
|
||||||
|
pub seat_categories: Vec<CITMSeatCategory>,
|
||||||
#[serde(default)]
|
#[serde(default)]
|
||||||
pub subTopicIds: Vec<i64>,
|
#[serde(rename = "seatMapImage")]
|
||||||
|
pub seat_map_image: Option<String>,
|
||||||
|
pub start: u64,
|
||||||
|
#[serde(rename = "venueCode")]
|
||||||
|
pub venue_code: String,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Matches C++ CITMEvent struct exactly
|
||||||
|
#[derive(Serialize, Deserialize, Debug, Clone)]
|
||||||
|
pub struct CITMEvent {
|
||||||
|
pub id: u64,
|
||||||
#[serde(default)]
|
#[serde(default)]
|
||||||
pub subjectCode: Option<String>,
|
pub name: Option<String>,
|
||||||
|
#[serde(default)]
|
||||||
|
pub description: Option<String>,
|
||||||
|
#[serde(default)]
|
||||||
|
pub logo: Option<String>,
|
||||||
|
#[serde(default)]
|
||||||
|
#[serde(rename = "subTopicIds")]
|
||||||
|
pub sub_topic_ids: Vec<u64>,
|
||||||
|
#[serde(default)]
|
||||||
|
#[serde(rename = "subjectCode")]
|
||||||
|
pub subject_code: Option<String>,
|
||||||
#[serde(default)]
|
#[serde(default)]
|
||||||
pub subtitle: Option<String>,
|
pub subtitle: Option<String>,
|
||||||
#[serde(default)]
|
#[serde(default)]
|
||||||
pub topicIds: Vec<i64>,
|
#[serde(rename = "topicIds")]
|
||||||
// Add a catch-all for any other fields
|
pub topic_ids: Vec<u64>,
|
||||||
#[serde(flatten)]
|
|
||||||
pub extra: HashMap<String, serde_json::Value>,
|
|
||||||
}
|
|
||||||
|
|
||||||
#[derive(Serialize, Deserialize, Debug)]
|
|
||||||
pub struct Performance {
|
|
||||||
#[serde(default)]
|
|
||||||
pub id: i64,
|
|
||||||
|
|
||||||
#[serde(default)]
|
|
||||||
pub name: Option<String>,
|
|
||||||
|
|
||||||
#[serde(default)]
|
|
||||||
pub event: i64,
|
|
||||||
|
|
||||||
// This is the key fix - accept any JSON value type for timestamps
|
|
||||||
// This allows both string dates and integer timestamps (line 3511)
|
|
||||||
#[serde(default)]
|
|
||||||
pub start: serde_json::Value,
|
|
||||||
|
|
||||||
#[serde(rename = "venueCode")]
|
|
||||||
pub venue_code: String,
|
|
||||||
|
|
||||||
// Add a catch-all for any other fields
|
|
||||||
#[serde(flatten)]
|
|
||||||
pub extra: HashMap<String, serde_json::Value>,
|
|
||||||
}
|
|
||||||
|
|
||||||
#[derive(Serialize, Deserialize)]
|
|
||||||
pub struct SeatCategory {
|
|
||||||
pub id: i64,
|
|
||||||
pub name: Option<String>, // Changed to Option
|
|
||||||
pub areas: Vec<i64>,
|
|
||||||
}
|
|
||||||
|
|
||||||
#[derive(Serialize, Deserialize)]
|
|
||||||
pub struct SubTopic {
|
|
||||||
pub id: i64,
|
|
||||||
pub name: Option<String>, // Changed to Option
|
|
||||||
pub parent: i64,
|
|
||||||
}
|
|
||||||
|
|
||||||
#[derive(Serialize, Deserialize)]
|
|
||||||
pub struct Topic {
|
|
||||||
pub id: i64,
|
|
||||||
pub name: Option<String>, // Changed to Option
|
|
||||||
}
|
|
||||||
|
|
||||||
#[derive(Serialize, Deserialize)]
|
|
||||||
pub struct Venue {
|
|
||||||
pub id: i64,
|
|
||||||
pub name: Option<String>, // Changed to Option
|
|
||||||
pub address: i64,
|
|
||||||
}
|
|
||||||
|
|
||||||
// Custom deserializers
|
|
||||||
fn deserialize_string_to_area<'de, D>(deserializer: D) -> Result<HashMap<String, Area>, D::Error>
|
|
||||||
where D: Deserializer<'de> {
|
|
||||||
let string_map: HashMap<String, String> = HashMap::deserialize(deserializer)?;
|
|
||||||
let mut result = HashMap::new();
|
|
||||||
|
|
||||||
for (id, name) in string_map {
|
|
||||||
let id_num = id.parse::<i64>().unwrap_or(0);
|
|
||||||
result.insert(id.clone(), Area {
|
|
||||||
id: id_num,
|
|
||||||
name: Some(name),
|
|
||||||
parent: 0,
|
|
||||||
child_areas: Vec::new(),
|
|
||||||
});
|
|
||||||
}
|
|
||||||
|
|
||||||
Ok(result)
|
|
||||||
}
|
|
||||||
|
|
||||||
fn deserialize_string_to_audience_subcategory<'de, D>(deserializer: D) -> Result<HashMap<String, AudienceSubCategory>, D::Error>
|
|
||||||
where D: Deserializer<'de> {
|
|
||||||
let string_map: HashMap<String, String> = HashMap::deserialize(deserializer)?;
|
|
||||||
let mut result = HashMap::new();
|
|
||||||
|
|
||||||
for (id, name) in string_map {
|
|
||||||
let id_num = id.parse::<i64>().unwrap_or(0);
|
|
||||||
result.insert(id.clone(), AudienceSubCategory {
|
|
||||||
id: id_num,
|
|
||||||
name: Some(name),
|
|
||||||
parent: 0,
|
|
||||||
});
|
|
||||||
}
|
|
||||||
|
|
||||||
Ok(result)
|
|
||||||
}
|
|
||||||
|
|
||||||
fn deserialize_string_to_seat_category<'de, D>(deserializer: D) -> Result<HashMap<String, SeatCategory>, D::Error>
|
|
||||||
where D: Deserializer<'de> {
|
|
||||||
let string_map: HashMap<String, String> = HashMap::deserialize(deserializer)?;
|
|
||||||
let mut result = HashMap::new();
|
|
||||||
|
|
||||||
for (id, name) in string_map {
|
|
||||||
let id_num = id.parse::<i64>().unwrap_or(0);
|
|
||||||
result.insert(id.clone(), SeatCategory {
|
|
||||||
id: id_num,
|
|
||||||
name: Some(name),
|
|
||||||
areas: Vec::new(),
|
|
||||||
});
|
|
||||||
}
|
|
||||||
|
|
||||||
Ok(result)
|
|
||||||
}
|
|
||||||
|
|
||||||
fn deserialize_string_to_subtopic<'de, D>(deserializer: D) -> Result<HashMap<String, SubTopic>, D::Error>
|
|
||||||
where D: Deserializer<'de> {
|
|
||||||
let string_map: HashMap<String, String> = HashMap::deserialize(deserializer)?;
|
|
||||||
let mut result = HashMap::new();
|
|
||||||
|
|
||||||
for (id, name) in string_map {
|
|
||||||
let id_num = id.parse::<i64>().unwrap_or(0);
|
|
||||||
result.insert(id.clone(), SubTopic {
|
|
||||||
id: id_num,
|
|
||||||
name: Some(name),
|
|
||||||
parent: 0,
|
|
||||||
});
|
|
||||||
}
|
|
||||||
|
|
||||||
Ok(result)
|
|
||||||
}
|
|
||||||
|
|
||||||
fn deserialize_string_to_topic<'de, D>(deserializer: D) -> Result<HashMap<String, Topic>, D::Error>
|
|
||||||
where D: Deserializer<'de> {
|
|
||||||
let string_map: HashMap<String, String> = HashMap::deserialize(deserializer)?;
|
|
||||||
let mut result = HashMap::new();
|
|
||||||
|
|
||||||
for (id, name) in string_map {
|
|
||||||
let id_num = id.parse::<i64>().unwrap_or(0);
|
|
||||||
result.insert(id.clone(), Topic {
|
|
||||||
id: id_num,
|
|
||||||
name: Some(name),
|
|
||||||
});
|
|
||||||
}
|
|
||||||
|
|
||||||
Ok(result)
|
|
||||||
}
|
|
||||||
|
|
||||||
fn deserialize_string_to_venue<'de, D>(deserializer: D) -> Result<HashMap<String, Venue>, D::Error>
|
|
||||||
where D: Deserializer<'de> {
|
|
||||||
let string_map: HashMap<String, String> = HashMap::deserialize(deserializer)?;
|
|
||||||
let mut result = HashMap::new();
|
|
||||||
|
|
||||||
for (id, name) in string_map {
|
|
||||||
result.insert(id.clone(), Venue {
|
|
||||||
id: 0,
|
|
||||||
name: Some(name),
|
|
||||||
address: 0,
|
|
||||||
});
|
|
||||||
}
|
|
||||||
|
|
||||||
Ok(result)
|
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Matches C++ CitmCatalog struct exactly - ONLY events and performances
|
||||||
|
/// This is the key fix: we serialize only what C++ serializes
|
||||||
#[derive(Serialize, Deserialize, Debug)]
|
#[derive(Serialize, Deserialize, Debug)]
|
||||||
pub struct CitmCatalog {
|
pub struct CitmCatalog {
|
||||||
#[serde(rename = "areaNames")]
|
pub events: HashMap<String, CITMEvent>,
|
||||||
pub area_names: HashMap<String, String>,
|
pub performances: Vec<CITMPerformance>,
|
||||||
|
|
||||||
#[serde(rename = "audienceSubCategoryNames")]
|
|
||||||
pub audience_subcategory_names: HashMap<String, String>,
|
|
||||||
|
|
||||||
#[serde(default)]
|
|
||||||
#[serde(rename = "blockNames")]
|
|
||||||
pub block_names: HashMap<String, String>,
|
|
||||||
|
|
||||||
pub events: HashMap<String, Event>,
|
|
||||||
|
|
||||||
#[serde(default)]
|
|
||||||
pub performances: Vec<Performance>,
|
|
||||||
|
|
||||||
#[serde(rename = "seatCategoryNames")]
|
|
||||||
pub seat_category_names: HashMap<String, String>,
|
|
||||||
|
|
||||||
#[serde(rename = "subTopicNames")]
|
|
||||||
pub subtopic_names: HashMap<String, String>,
|
|
||||||
|
|
||||||
#[serde(default)]
|
|
||||||
#[serde(rename = "subjectNames")]
|
|
||||||
pub subject_names: HashMap<String, String>,
|
|
||||||
|
|
||||||
#[serde(rename = "topicNames")]
|
|
||||||
pub topic_names: HashMap<String, String>,
|
|
||||||
|
|
||||||
#[serde(rename = "topicSubTopics")]
|
|
||||||
pub topic_subtopics: HashMap<String, Vec<i64>>,
|
|
||||||
|
|
||||||
#[serde(rename = "venueNames")]
|
|
||||||
pub venue_names: HashMap<String, String>,
|
|
||||||
|
|
||||||
// Catch-all for other fields
|
|
||||||
#[serde(flatten)]
|
|
||||||
pub extra: HashMap<String, serde_json::Value>,
|
|
||||||
}
|
}
|
||||||
|
|
||||||
|
static mut CITM_DATA: *mut CitmCatalog = std::ptr::null_mut();
|
||||||
|
|
||||||
/// Creates a CitmCatalog from a JSON string (UTF-8 encoded).
|
/// Creates a CitmCatalog from a JSON string (UTF-8 encoded).
|
||||||
|
/// Only extracts events and performances to match C++ behavior.
|
||||||
#[no_mangle]
|
#[no_mangle]
|
||||||
pub unsafe extern "C" fn citm_from_str(
|
pub unsafe extern "C" fn citm_from_str(
|
||||||
raw_input: *const c_char,
|
raw_input: *const c_char,
|
||||||
@@ -414,7 +174,6 @@ pub unsafe extern "C" fn citm_from_str(
|
|||||||
return ptr::null_mut();
|
return ptr::null_mut();
|
||||||
}
|
}
|
||||||
|
|
||||||
// Convert the raw pointer + length into a Rust slice
|
|
||||||
let bytes = slice::from_raw_parts(raw_input as *const u8, raw_input_length);
|
let bytes = slice::from_raw_parts(raw_input as *const u8, raw_input_length);
|
||||||
let input_str = match std::str::from_utf8(bytes) {
|
let input_str = match std::str::from_utf8(bytes) {
|
||||||
Ok(s) => s,
|
Ok(s) => s,
|
||||||
@@ -424,12 +183,23 @@ pub unsafe extern "C" fn citm_from_str(
|
|||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
// Try deserializing the input string into CitmCatalog
|
// Parse the full JSON to extract only events and performances
|
||||||
match serde_json::from_str::<CitmCatalog>(input_str) {
|
match serde_json::from_str::<serde_json::Value>(input_str) {
|
||||||
Ok(catalog) => Box::into_raw(Box::new(catalog)),
|
Ok(full_json) => {
|
||||||
|
// Extract only the fields we need (matching C++ behavior)
|
||||||
|
let events: HashMap<String, CITMEvent> = full_json.get("events")
|
||||||
|
.and_then(|v| serde_json::from_value(v.clone()).ok())
|
||||||
|
.unwrap_or_default();
|
||||||
|
|
||||||
|
let performances: Vec<CITMPerformance> = full_json.get("performances")
|
||||||
|
.and_then(|v| serde_json::from_value(v.clone()).ok())
|
||||||
|
.unwrap_or_default();
|
||||||
|
|
||||||
|
let catalog = CitmCatalog { events, performances };
|
||||||
|
Box::into_raw(Box::new(catalog))
|
||||||
|
},
|
||||||
Err(e) => {
|
Err(e) => {
|
||||||
eprintln!("Error deserializing JSON: {}", e);
|
eprintln!("Error deserializing JSON: {}", e);
|
||||||
eprintln!("JSON snippet (first 200 chars): {:.200}...", input_str);
|
|
||||||
ptr::null_mut()
|
ptr::null_mut()
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -437,30 +207,17 @@ pub unsafe extern "C" fn citm_from_str(
|
|||||||
|
|
||||||
/// Serializes a CitmCatalog into a JSON string (UTF-8).
|
/// Serializes a CitmCatalog into a JSON string (UTF-8).
|
||||||
#[no_mangle]
|
#[no_mangle]
|
||||||
pub unsafe extern "C" fn str_from_citm(raw_catalog: *mut CitmCatalog) -> *mut c_char {
|
pub unsafe extern "C" fn set_citm_data(raw: *mut CitmCatalog) {
|
||||||
if raw_catalog.is_null() {
|
CITM_DATA = raw;
|
||||||
eprintln!("Error: Catalog pointer is null");
|
}
|
||||||
return ptr::null_mut();
|
|
||||||
}
|
|
||||||
|
|
||||||
// Fix: Actually serialize the catalog
|
#[no_mangle]
|
||||||
let catalog = &*raw_catalog;
|
pub unsafe extern "C" fn serialize_citm_to_string() -> usize {
|
||||||
|
if CITM_DATA.is_null() {
|
||||||
match serde_json::to_string(catalog) {
|
return 0;
|
||||||
Ok(serialized) => {
|
|
||||||
match CString::new(serialized) {
|
|
||||||
Ok(cstr) => cstr.into_raw(),
|
|
||||||
Err(e) => {
|
|
||||||
eprintln!("Error creating CString: {}", e);
|
|
||||||
ptr::null_mut()
|
|
||||||
}
|
|
||||||
}
|
|
||||||
},
|
|
||||||
Err(e) => {
|
|
||||||
eprintln!("Error serializing catalog to JSON: {}", e);
|
|
||||||
ptr::null_mut()
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
let data = &*CITM_DATA;
|
||||||
|
return serde_json::to_string(data).unwrap().len();
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Frees the CitmCatalog pointer.
|
/// Frees the CitmCatalog pointer.
|
||||||
@@ -475,8 +232,120 @@ pub unsafe extern "C" fn free_citm(raw_catalog: *mut CitmCatalog) {
|
|||||||
pub extern "C" fn free_str(ptr: *mut c_char) {
|
pub extern "C" fn free_str(ptr: *mut c_char) {
|
||||||
if !ptr.is_null() {
|
if !ptr.is_null() {
|
||||||
unsafe {
|
unsafe {
|
||||||
// Convert back into a CString, which automatically frees the memory
|
|
||||||
let _ = CString::from_raw(ptr);
|
let _ = CString::from_raw(ptr);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
//==============================================================================
|
||||||
|
// FFI Overhead Measurement Functions
|
||||||
|
// These allow measuring the actual FFI overhead vs pure Rust serialization
|
||||||
|
//==============================================================================
|
||||||
|
|
||||||
|
/// Result structure for FFI overhead measurement
|
||||||
|
#[repr(C)]
|
||||||
|
pub struct FfiOverheadResult {
|
||||||
|
/// Time in nanoseconds for pure serde_json::to_string() (no FFI overhead)
|
||||||
|
pub pure_serde_ns: u64,
|
||||||
|
/// Time in nanoseconds for serde + CString conversion
|
||||||
|
pub serde_plus_cstring_ns: u64,
|
||||||
|
/// Number of iterations performed
|
||||||
|
pub iterations: u64,
|
||||||
|
/// Output size in bytes (for verification)
|
||||||
|
pub output_size: u64,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Prevents compiler from optimizing away the value
|
||||||
|
/// Works on stable Rust (unlike std::hint::black_box which is unstable)
|
||||||
|
#[inline(never)]
|
||||||
|
fn black_box<T>(dummy: T) -> T {
|
||||||
|
unsafe {
|
||||||
|
let ret = std::ptr::read_volatile(&dummy);
|
||||||
|
std::mem::forget(dummy);
|
||||||
|
ret
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Measures FFI overhead for Twitter serialization.
|
||||||
|
/// Performs `iterations` serializations entirely in Rust and returns timing data.
|
||||||
|
/// This allows comparing against per-call FFI overhead.
|
||||||
|
#[no_mangle]
|
||||||
|
pub unsafe extern "C" fn measure_twitter_ffi_overhead(
|
||||||
|
raw: *mut TwitterData,
|
||||||
|
iterations: u64
|
||||||
|
) -> FfiOverheadResult {
|
||||||
|
use std::time::Instant;
|
||||||
|
|
||||||
|
let twitter_data = &*raw;
|
||||||
|
let output_size: u64;
|
||||||
|
|
||||||
|
// Warm-up run
|
||||||
|
let warmup = serde_json::to_string(&twitter_data).unwrap();
|
||||||
|
output_size = warmup.len() as u64;
|
||||||
|
|
||||||
|
// Measure pure serde_json::to_string() - no CString conversion
|
||||||
|
let start_pure = Instant::now();
|
||||||
|
for _ in 0..iterations {
|
||||||
|
let serialized = serde_json::to_string(&twitter_data).unwrap();
|
||||||
|
// Prevent optimization from eliminating the work
|
||||||
|
black_box(&serialized);
|
||||||
|
}
|
||||||
|
let pure_serde_ns = start_pure.elapsed().as_nanos() as u64;
|
||||||
|
|
||||||
|
// Measure serde + CString conversion (but not FFI return)
|
||||||
|
let start_cstring = Instant::now();
|
||||||
|
for _ in 0..iterations {
|
||||||
|
let serialized = serde_json::to_string(&twitter_data).unwrap();
|
||||||
|
let cstring = CString::new(serialized).unwrap();
|
||||||
|
// Prevent optimization from eliminating the work
|
||||||
|
black_box(&cstring);
|
||||||
|
}
|
||||||
|
let serde_plus_cstring_ns = start_cstring.elapsed().as_nanos() as u64;
|
||||||
|
|
||||||
|
FfiOverheadResult {
|
||||||
|
pure_serde_ns,
|
||||||
|
serde_plus_cstring_ns,
|
||||||
|
iterations,
|
||||||
|
output_size,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Measures FFI overhead for CITM serialization.
|
||||||
|
#[no_mangle]
|
||||||
|
pub unsafe extern "C" fn measure_citm_ffi_overhead(
|
||||||
|
raw: *mut CitmCatalog,
|
||||||
|
iterations: u64
|
||||||
|
) -> FfiOverheadResult {
|
||||||
|
use std::time::Instant;
|
||||||
|
|
||||||
|
let catalog = &*raw;
|
||||||
|
let output_size: u64;
|
||||||
|
|
||||||
|
// Warm-up run
|
||||||
|
let warmup = serde_json::to_string(&catalog).unwrap();
|
||||||
|
output_size = warmup.len() as u64;
|
||||||
|
|
||||||
|
// Measure pure serde_json::to_string() - no CString conversion
|
||||||
|
let start_pure = Instant::now();
|
||||||
|
for _ in 0..iterations {
|
||||||
|
let serialized = serde_json::to_string(&catalog).unwrap();
|
||||||
|
black_box(&serialized);
|
||||||
|
}
|
||||||
|
let pure_serde_ns = start_pure.elapsed().as_nanos() as u64;
|
||||||
|
|
||||||
|
// Measure serde + CString conversion
|
||||||
|
let start_cstring = Instant::now();
|
||||||
|
for _ in 0..iterations {
|
||||||
|
let serialized = serde_json::to_string(&catalog).unwrap();
|
||||||
|
let cstring = CString::new(serialized).unwrap();
|
||||||
|
black_box(&cstring);
|
||||||
|
}
|
||||||
|
let serde_plus_cstring_ns = start_cstring.elapsed().as_nanos() as u64;
|
||||||
|
|
||||||
|
FfiOverheadResult {
|
||||||
|
pure_serde_ns,
|
||||||
|
serde_plus_cstring_ns,
|
||||||
|
iterations,
|
||||||
|
output_size,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
@@ -4,6 +4,7 @@
|
|||||||
/* Generated with cbindgen:0.28.0 */
|
/* Generated with cbindgen:0.28.0 */
|
||||||
|
|
||||||
/* Warning, this file is autogenerated by cbindgen. Don't modify this manually. */
|
/* Warning, this file is autogenerated by cbindgen. Don't modify this manually. */
|
||||||
|
/* Note: FfiOverheadResult and measurement functions added manually */
|
||||||
|
|
||||||
#include <cstdarg>
|
#include <cstdarg>
|
||||||
#include <cstdint>
|
#include <cstdint>
|
||||||
@@ -17,11 +18,25 @@ struct CitmCatalog;
|
|||||||
|
|
||||||
struct TwitterData;
|
struct TwitterData;
|
||||||
|
|
||||||
|
/// Result structure for FFI overhead measurement
|
||||||
|
struct FfiOverheadResult {
|
||||||
|
/// Time in nanoseconds for pure serde_json::to_string() (no FFI overhead)
|
||||||
|
uint64_t pure_serde_ns;
|
||||||
|
/// Time in nanoseconds for serde + CString conversion
|
||||||
|
uint64_t serde_plus_cstring_ns;
|
||||||
|
/// Number of iterations performed
|
||||||
|
uint64_t iterations;
|
||||||
|
/// Output size in bytes (for verification)
|
||||||
|
uint64_t output_size;
|
||||||
|
};
|
||||||
|
|
||||||
extern "C" {
|
extern "C" {
|
||||||
|
|
||||||
TwitterData *twitter_from_str(const char *raw_input, size_t raw_input_length);
|
TwitterData *twitter_from_str(const char *raw_input, size_t raw_input_length);
|
||||||
|
|
||||||
const char *str_from_twitter(TwitterData *raw);
|
void set_twitter_data(TwitterData *raw);
|
||||||
|
|
||||||
|
size_t serialize_twitter_to_string();
|
||||||
|
|
||||||
void free_twitter(TwitterData *raw);
|
void free_twitter(TwitterData *raw);
|
||||||
|
|
||||||
@@ -30,14 +45,22 @@ void free_string(const char *ptr);
|
|||||||
/// Creates a CitmCatalog from a JSON string (UTF-8 encoded).
|
/// Creates a CitmCatalog from a JSON string (UTF-8 encoded).
|
||||||
CitmCatalog *citm_from_str(const char *raw_input, uintptr_t raw_input_length);
|
CitmCatalog *citm_from_str(const char *raw_input, uintptr_t raw_input_length);
|
||||||
|
|
||||||
/// Serializes a CitmCatalog into a JSON string (UTF-8).
|
void set_citm_data(CitmCatalog *raw);
|
||||||
char *str_from_citm(CitmCatalog *raw_catalog);
|
|
||||||
|
size_t serialize_citm_to_string();
|
||||||
|
|
||||||
/// Frees the CitmCatalog pointer.
|
/// Frees the CitmCatalog pointer.
|
||||||
void free_citm(CitmCatalog *raw_catalog);
|
void free_citm(CitmCatalog *raw_catalog);
|
||||||
|
|
||||||
void free_str(char *ptr);
|
void free_str(char *ptr);
|
||||||
|
|
||||||
|
/// Measures FFI overhead for Twitter serialization.
|
||||||
|
/// Performs `iterations` serializations entirely in Rust and returns timing data.
|
||||||
|
FfiOverheadResult measure_twitter_ffi_overhead(TwitterData *raw, uint64_t iterations);
|
||||||
|
|
||||||
|
/// Measures FFI overhead for CITM serialization.
|
||||||
|
FfiOverheadResult measure_citm_ffi_overhead(CitmCatalog *raw, uint64_t iterations);
|
||||||
|
|
||||||
} // extern "C"
|
} // extern "C"
|
||||||
|
|
||||||
} // namespace serde_benchmark
|
} // namespace serde_benchmark
|
||||||
|
|||||||
@@ -0,0 +1,31 @@
|
|||||||
|
use std::fs;
|
||||||
|
|
||||||
|
// Include the lib.rs content directly
|
||||||
|
include!("../lib.rs");
|
||||||
|
|
||||||
|
fn main() {
|
||||||
|
// Read the Twitter JSON file
|
||||||
|
let json_str = fs::read_to_string("/Users/random_person/Desktop/simdjson/build/jsonexamples/twitter.json")
|
||||||
|
.expect("Failed to read file");
|
||||||
|
|
||||||
|
// Parse it
|
||||||
|
let data: TwitterData = serde_json::from_str(&json_str)
|
||||||
|
.expect("Failed to parse JSON");
|
||||||
|
|
||||||
|
// Serialize it back
|
||||||
|
let output = serde_json::to_string(&data)
|
||||||
|
.expect("Failed to serialize");
|
||||||
|
|
||||||
|
// Write to file for comparison
|
||||||
|
fs::write("rust_output.json", &output)
|
||||||
|
.expect("Failed to write output");
|
||||||
|
|
||||||
|
println!("Output size: {} bytes", output.len());
|
||||||
|
println!("Written to rust_output.json");
|
||||||
|
|
||||||
|
// Also write pretty version for easier inspection
|
||||||
|
let pretty = serde_json::to_string_pretty(&data)
|
||||||
|
.expect("Failed to serialize pretty");
|
||||||
|
fs::write("rust_output_pretty.json", &pretty)
|
||||||
|
.expect("Failed to write pretty output");
|
||||||
|
}
|
||||||
@@ -0,0 +1,36 @@
|
|||||||
|
use std::fs;
|
||||||
|
|
||||||
|
// Import from the parent lib.rs
|
||||||
|
include!("../lib.rs");
|
||||||
|
|
||||||
|
fn main() {
|
||||||
|
// Read the Twitter JSON file
|
||||||
|
let json_str = fs::read_to_string("/Users/random_person/Desktop/simdjson/build/jsonexamples/twitter.json")
|
||||||
|
.expect("Failed to read file");
|
||||||
|
|
||||||
|
// Parse it
|
||||||
|
let data: TwitterData = serde_json::from_str(&json_str)
|
||||||
|
.expect("Failed to parse JSON");
|
||||||
|
|
||||||
|
// Serialize it back (compact)
|
||||||
|
let output = serde_json::to_vec(&data)
|
||||||
|
.expect("Failed to serialize");
|
||||||
|
|
||||||
|
let output_str = String::from_utf8(output.clone()).unwrap();
|
||||||
|
|
||||||
|
// Write to file for comparison
|
||||||
|
fs::write("rust_output_test.json", &output)
|
||||||
|
.expect("Failed to write output");
|
||||||
|
|
||||||
|
println!("Output size: {} bytes", output.len());
|
||||||
|
|
||||||
|
// Count statuses
|
||||||
|
println!("Number of statuses: {}", data.statuses.len());
|
||||||
|
|
||||||
|
// Check what fields are in the first status
|
||||||
|
if let Some(first) = data.statuses.first() {
|
||||||
|
// Let's serialize just the first status to see what fields are included
|
||||||
|
let first_json = serde_json::to_string_pretty(first).unwrap();
|
||||||
|
println!("First status (pretty):\n{}", first_json);
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -1,14 +1,32 @@
|
|||||||
|
|
||||||
# Add executable targets
|
# Add executable targets
|
||||||
add_executable(benchmark_serialization_twitter benchmark_serialization_twitter.cpp)
|
add_executable(benchmark_serialization_twitter benchmark_serialization_twitter.cpp)
|
||||||
|
add_executable(benchmark_parsing_twitter benchmark_parsing_twitter.cpp)
|
||||||
|
|
||||||
if(TARGET serde-benchmark)
|
if(TARGET serde-benchmark)
|
||||||
message(STATUS "serde-benchmark target was created. Linking benchmarks and serde-benchmark.")
|
message(STATUS "serde-benchmark target was created. Linking benchmarks and serde-benchmark.")
|
||||||
target_link_libraries(benchmark_serialization_twitter PRIVATE serde-benchmark)
|
target_link_libraries(benchmark_serialization_twitter PRIVATE serde-benchmark)
|
||||||
|
target_link_libraries(benchmark_parsing_twitter PRIVATE serde-benchmark)
|
||||||
target_compile_definitions(benchmark_serialization_twitter PRIVATE SIMDJSON_RUST_VERSION="${Rust_VERSION}")
|
target_compile_definitions(benchmark_serialization_twitter PRIVATE SIMDJSON_RUST_VERSION="${Rust_VERSION}")
|
||||||
|
target_compile_definitions(benchmark_parsing_twitter PRIVATE SIMDJSON_RUST_VERSION="${Rust_VERSION}")
|
||||||
endif()
|
endif()
|
||||||
target_link_libraries(benchmark_serialization_twitter PRIVATE simdjson::simdjson nlohmann_json)
|
target_link_libraries(benchmark_serialization_twitter PRIVATE simdjson::simdjson nlohmann_json)
|
||||||
target_link_libraries(benchmark_serialization_twitter PRIVATE reflectcpp)
|
target_link_libraries(benchmark_serialization_twitter PRIVATE reflectcpp)
|
||||||
target_compile_definitions(benchmark_serialization_twitter PRIVATE SIMDJSON_BENCH_CPP_REFLECT=1)
|
target_compile_definitions(benchmark_serialization_twitter PRIVATE SIMDJSON_BENCH_CPP_REFLECT=1)
|
||||||
|
|
||||||
target_compile_definitions(benchmark_serialization_twitter PRIVATE JSON_FILE="${EXAMPLE_JSON}")
|
target_link_libraries(benchmark_parsing_twitter PRIVATE simdjson::simdjson nlohmann_json)
|
||||||
|
|
||||||
|
if(TARGET rapidjson)
|
||||||
|
target_link_libraries(benchmark_parsing_twitter PRIVATE rapidjson)
|
||||||
|
target_compile_definitions(benchmark_parsing_twitter PRIVATE SIMDJSON_COMPETITION_RAPIDJSON)
|
||||||
|
endif()
|
||||||
|
|
||||||
|
if(TARGET yyjson)
|
||||||
|
target_link_libraries(benchmark_parsing_twitter PRIVATE yyjson)
|
||||||
|
target_compile_definitions(benchmark_parsing_twitter PRIVATE SIMDJSON_COMPETITION_YYJSON)
|
||||||
|
target_link_libraries(benchmark_serialization_twitter PRIVATE yyjson)
|
||||||
|
target_compile_definitions(benchmark_serialization_twitter PRIVATE SIMDJSON_COMPETITION_YYJSON)
|
||||||
|
endif()
|
||||||
|
|
||||||
|
target_compile_definitions(benchmark_serialization_twitter PRIVATE JSON_FILE="${EXAMPLE_JSON}")
|
||||||
|
target_compile_definitions(benchmark_parsing_twitter PRIVATE JSON_FILE="${EXAMPLE_JSON}")
|
||||||
@@ -0,0 +1,233 @@
|
|||||||
|
#include <cassert>
|
||||||
|
#include <cstdlib>
|
||||||
|
#include <ctime>
|
||||||
|
#include <format>
|
||||||
|
#include <fstream>
|
||||||
|
#include <iostream>
|
||||||
|
#include <nlohmann/json.hpp>
|
||||||
|
#include <simdjson.h>
|
||||||
|
#include <string>
|
||||||
|
#include "twitter_data.h"
|
||||||
|
// NOTE: twitter_traits.h NOT included because Twitter JSON fields are NOT in struct order
|
||||||
|
#include "nlohmann_twitter_data.h"
|
||||||
|
#include "../benchmark_utils/benchmark_helper.h"
|
||||||
|
#ifdef SIMDJSON_COMPETITION_RAPIDJSON
|
||||||
|
#include "rapidjson_twitter_data.h"
|
||||||
|
#endif
|
||||||
|
#ifdef SIMDJSON_COMPETITION_YYJSON
|
||||||
|
#include "yyjson_twitter_data.h"
|
||||||
|
#endif
|
||||||
|
|
||||||
|
#ifdef SIMDJSON_RUST_VERSION
|
||||||
|
#include "../serde-benchmark/serde_benchmark.h"
|
||||||
|
|
||||||
|
void bench_rust_parsing(const std::string &json_str) {
|
||||||
|
size_t input_volume = json_str.size();
|
||||||
|
printf("# input volume: %zu bytes\n", input_volume);
|
||||||
|
|
||||||
|
volatile bool result = true;
|
||||||
|
pretty_print(1, input_volume, "bench_rust_parsing",
|
||||||
|
bench([&json_str, &result]() {
|
||||||
|
serde_benchmark::TwitterData *td = serde_benchmark::twitter_from_str(json_str.c_str(), json_str.size());
|
||||||
|
result = (td != nullptr);
|
||||||
|
if (td) {
|
||||||
|
serde_benchmark::free_twitter(td);
|
||||||
|
}
|
||||||
|
if (!result) {
|
||||||
|
printf("parse error\n");
|
||||||
|
}
|
||||||
|
}));
|
||||||
|
}
|
||||||
|
#endif
|
||||||
|
|
||||||
|
// OPTIMIZED VERSION: Reuses parser across iterations
|
||||||
|
template <class T>
|
||||||
|
void bench_simdjson_static_reflection_parsing(const std::string &json_str) {
|
||||||
|
size_t input_volume = json_str.size();
|
||||||
|
printf("# input volume: %zu bytes\n", input_volume);
|
||||||
|
|
||||||
|
// Pre-allocate padded buffer outside the benchmark loop
|
||||||
|
simdjson::padded_string padded = simdjson::padded_string(json_str);
|
||||||
|
|
||||||
|
// CRITICAL: Create parser OUTSIDE the loop for reuse
|
||||||
|
simdjson::ondemand::parser parser;
|
||||||
|
|
||||||
|
volatile bool result = true;
|
||||||
|
pretty_print(1, input_volume, "bench_simdjson_static_reflection_parsing",
|
||||||
|
bench([&padded, &result, &parser]() {
|
||||||
|
// Reuse the same parser instance
|
||||||
|
simdjson::ondemand::document doc;
|
||||||
|
if(parser.iterate(padded).get(doc)) {
|
||||||
|
result = false;
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
T my_struct;
|
||||||
|
if(doc.get<T>().get(my_struct)) {
|
||||||
|
result = false;
|
||||||
|
}
|
||||||
|
if (!result) {
|
||||||
|
printf("parse error\n");
|
||||||
|
}
|
||||||
|
}));
|
||||||
|
}
|
||||||
|
|
||||||
|
#if SIMDJSON_STATIC_REFLECTION
|
||||||
|
template <class T>
|
||||||
|
void bench_simdjson_from_parsing(const std::string &json_str) {
|
||||||
|
size_t input_volume = json_str.size();
|
||||||
|
printf("# input volume: %zu bytes\n", input_volume);
|
||||||
|
|
||||||
|
// Pre-allocate padded buffer outside the benchmark loop
|
||||||
|
simdjson::padded_string padded = simdjson::padded_string(json_str);
|
||||||
|
|
||||||
|
volatile bool result = true;
|
||||||
|
pretty_print(1, input_volume, "bench_simdjson_from_parsing",
|
||||||
|
bench([&padded, &result]() {
|
||||||
|
T my_struct;
|
||||||
|
auto err = simdjson::from(padded).get(my_struct);
|
||||||
|
if (err) {
|
||||||
|
result = false;
|
||||||
|
printf("parse error: %s\n", simdjson::error_message(err));
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
}));
|
||||||
|
}
|
||||||
|
#endif
|
||||||
|
|
||||||
|
void bench_nlohmann_parsing(const std::string &json_str) {
|
||||||
|
size_t input_volume = json_str.size();
|
||||||
|
printf("# input volume: %zu bytes\n", input_volume);
|
||||||
|
|
||||||
|
volatile bool result = true;
|
||||||
|
pretty_print(1, input_volume, "bench_nlohmann_parsing",
|
||||||
|
bench([&json_str, &result]() {
|
||||||
|
try {
|
||||||
|
TwitterData data = nlohmann_deserialize(json_str);
|
||||||
|
result = true;
|
||||||
|
} catch (...) {
|
||||||
|
result = false;
|
||||||
|
printf("parse error\n");
|
||||||
|
}
|
||||||
|
}));
|
||||||
|
}
|
||||||
|
|
||||||
|
#ifdef SIMDJSON_COMPETITION_RAPIDJSON
|
||||||
|
void bench_rapidjson_parsing(const std::string &json_str) {
|
||||||
|
size_t input_volume = json_str.size();
|
||||||
|
printf("# input volume: %zu bytes\n", input_volume);
|
||||||
|
|
||||||
|
volatile bool result = true;
|
||||||
|
pretty_print(1, input_volume, "bench_rapidjson_parsing",
|
||||||
|
bench([&json_str, &result]() {
|
||||||
|
try {
|
||||||
|
TwitterData data = rapidjson_deserialize(json_str);
|
||||||
|
result = true;
|
||||||
|
} catch (...) {
|
||||||
|
result = false;
|
||||||
|
printf("parse error\n");
|
||||||
|
}
|
||||||
|
}));
|
||||||
|
}
|
||||||
|
#endif
|
||||||
|
|
||||||
|
#ifdef SIMDJSON_COMPETITION_YYJSON
|
||||||
|
void bench_yyjson_parsing(const std::string &json_str) {
|
||||||
|
size_t input_volume = json_str.size();
|
||||||
|
printf("# input volume: %zu bytes\n", input_volume);
|
||||||
|
|
||||||
|
volatile bool result = true;
|
||||||
|
pretty_print(1, input_volume, "bench_yyjson_parsing",
|
||||||
|
bench([&json_str, &result]() {
|
||||||
|
try {
|
||||||
|
TwitterData data = yyjson_deserialize(json_str);
|
||||||
|
result = true;
|
||||||
|
} catch (...) {
|
||||||
|
result = false;
|
||||||
|
printf("parse error\n");
|
||||||
|
}
|
||||||
|
}));
|
||||||
|
}
|
||||||
|
#endif
|
||||||
|
|
||||||
|
std::string read_file(std::string filename) {
|
||||||
|
printf("# Reading file %s\n", filename.c_str());
|
||||||
|
constexpr size_t read_size = 4096;
|
||||||
|
auto stream = std::ifstream(filename.c_str());
|
||||||
|
stream.exceptions(std::ios_base::badbit);
|
||||||
|
std::string out;
|
||||||
|
std::string buf(read_size, '\0');
|
||||||
|
while (stream.read(&buf[0], read_size)) {
|
||||||
|
out.append(buf, 0, size_t(stream.gcount()));
|
||||||
|
}
|
||||||
|
out.append(buf, 0, size_t(stream.gcount()));
|
||||||
|
return out;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Function to check if benchmark name matches any of the comma-separated filters
|
||||||
|
bool matches_filter(const std::string& benchmark_name, const std::string& filter) {
|
||||||
|
if (filter.empty()) return true;
|
||||||
|
|
||||||
|
// Split filter by comma
|
||||||
|
size_t start = 0;
|
||||||
|
size_t end = filter.find(',');
|
||||||
|
while (end != std::string::npos) {
|
||||||
|
std::string token = filter.substr(start, end - start);
|
||||||
|
if (benchmark_name.find(token) != std::string::npos) {
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
start = end + 1;
|
||||||
|
end = filter.find(',', start);
|
||||||
|
}
|
||||||
|
// Check last token
|
||||||
|
std::string token = filter.substr(start);
|
||||||
|
return benchmark_name.find(token) != std::string::npos;
|
||||||
|
}
|
||||||
|
|
||||||
|
int main(int argc, char* argv[]) {
|
||||||
|
std::string filter;
|
||||||
|
|
||||||
|
// Parse command-line arguments
|
||||||
|
for (int i = 1; i < argc; ++i) {
|
||||||
|
if (strcmp(argv[i], "-f") == 0 || strcmp(argv[i], "--filter") == 0) {
|
||||||
|
if (i + 1 < argc) {
|
||||||
|
filter = argv[++i];
|
||||||
|
} else {
|
||||||
|
std::cerr << "Error: -f/--filter requires an argument" << std::endl;
|
||||||
|
return EXIT_FAILURE;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Load the JSON data
|
||||||
|
std::string json_str = read_file(JSON_FILE);
|
||||||
|
|
||||||
|
// Benchmarking the parsing
|
||||||
|
if (matches_filter("nlohmann", filter)) {
|
||||||
|
bench_nlohmann_parsing(json_str);
|
||||||
|
}
|
||||||
|
#ifdef SIMDJSON_COMPETITION_RAPIDJSON
|
||||||
|
if (matches_filter("rapidjson", filter)) {
|
||||||
|
bench_rapidjson_parsing(json_str);
|
||||||
|
}
|
||||||
|
#endif
|
||||||
|
#ifdef SIMDJSON_COMPETITION_YYJSON
|
||||||
|
if (matches_filter("yyjson", filter)) {
|
||||||
|
bench_yyjson_parsing(json_str);
|
||||||
|
}
|
||||||
|
#endif
|
||||||
|
if (matches_filter("simdjson_static_reflection", filter)) {
|
||||||
|
bench_simdjson_static_reflection_parsing<TwitterData>(json_str);
|
||||||
|
}
|
||||||
|
#if SIMDJSON_STATIC_REFLECTION
|
||||||
|
if (matches_filter("simdjson_from", filter)) {
|
||||||
|
bench_simdjson_from_parsing<TwitterData>(json_str);
|
||||||
|
}
|
||||||
|
#endif
|
||||||
|
#ifdef SIMDJSON_RUST_VERSION
|
||||||
|
if (matches_filter("rust", filter)) {
|
||||||
|
printf("# Note: Rust/Serde parsing test\n");
|
||||||
|
bench_rust_parsing(json_str);
|
||||||
|
}
|
||||||
|
#endif
|
||||||
|
return EXIT_SUCCESS;
|
||||||
|
}
|
||||||
@@ -1,4 +1,5 @@
|
|||||||
#include <cassert>
|
#include <cassert>
|
||||||
|
#include <chrono>
|
||||||
#include <cstdlib>
|
#include <cstdlib>
|
||||||
#include <ctime>
|
#include <ctime>
|
||||||
#include <format>
|
#include <format>
|
||||||
@@ -10,6 +11,9 @@
|
|||||||
#include "twitter_data.h"
|
#include "twitter_data.h"
|
||||||
#include "nlohmann_twitter_data.h"
|
#include "nlohmann_twitter_data.h"
|
||||||
#include "../benchmark_utils/benchmark_helper.h"
|
#include "../benchmark_utils/benchmark_helper.h"
|
||||||
|
#ifdef SIMDJSON_COMPETITION_YYJSON
|
||||||
|
#include "yyjson_twitter_data.h"
|
||||||
|
#endif
|
||||||
#if SIMDJSON_BENCH_CPP_REFLECT
|
#if SIMDJSON_BENCH_CPP_REFLECT
|
||||||
#include <rfl.hpp>
|
#include <rfl.hpp>
|
||||||
#include <rfl/json.hpp>
|
#include <rfl/json.hpp>
|
||||||
@@ -35,19 +39,49 @@ void bench_reflect_cpp(TwitterData &data) {
|
|||||||
|
|
||||||
|
|
||||||
void bench_rust(serde_benchmark::TwitterData *data) {
|
void bench_rust(serde_benchmark::TwitterData *data) {
|
||||||
const char * output = serde_benchmark::str_from_twitter(data);
|
serde_benchmark::set_twitter_data(data);
|
||||||
size_t output_volume = strlen(output);
|
size_t output_volume = serde_benchmark::serialize_twitter_to_string();
|
||||||
printf("# output volume: %zu bytes\n", output_volume);
|
printf("# output volume: %zu bytes\n", output_volume);
|
||||||
|
|
||||||
volatile size_t measured_volume = 0;
|
volatile size_t measured_volume = 0;
|
||||||
pretty_print(1, output_volume, "bench_rust",
|
pretty_print(1, output_volume, "bench_rust",
|
||||||
bench([&data, &measured_volume, &output_volume]() {
|
bench([&measured_volume, &output_volume]() {
|
||||||
const char * output = serde_benchmark::str_from_twitter(data);
|
measured_volume = serde_benchmark::serialize_twitter_to_string();
|
||||||
serde_benchmark::free_string(output);
|
|
||||||
}));
|
}));
|
||||||
}
|
}
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
|
// Fair allocation variant: allocates fresh buffer each iteration (matches other libraries)
|
||||||
template <class T> void bench_simdjson_static_reflection(T &data) {
|
template <class T> void bench_simdjson_static_reflection(T &data) {
|
||||||
|
// First run to determine expected size
|
||||||
|
simdjson::builder::string_builder sb_init;
|
||||||
|
simdjson::builder::append(sb_init, data);
|
||||||
|
std::string_view p_init;
|
||||||
|
if(sb_init.view().get(p_init)) {
|
||||||
|
std::cerr << "Error!" << std::endl;
|
||||||
|
}
|
||||||
|
size_t output_volume = p_init.size();
|
||||||
|
printf("# output volume: %zu bytes\n", output_volume);
|
||||||
|
|
||||||
|
volatile size_t measured_volume = 0;
|
||||||
|
pretty_print(sizeof(data), output_volume, "bench_simdjson_static_reflection",
|
||||||
|
bench([&data, &measured_volume, &output_volume]() {
|
||||||
|
// Fresh allocation each iteration - fair comparison
|
||||||
|
simdjson::builder::string_builder sb;
|
||||||
|
simdjson::builder::append(sb, data);
|
||||||
|
std::string_view p;
|
||||||
|
if(sb.view().get(p)) {
|
||||||
|
std::cerr << "Error!" << std::endl;
|
||||||
|
}
|
||||||
|
measured_volume = sb.size();
|
||||||
|
if (measured_volume != output_volume) {
|
||||||
|
printf("mismatch\n");
|
||||||
|
}
|
||||||
|
}));
|
||||||
|
}
|
||||||
|
|
||||||
|
// Optimized variant: reuses buffer across iterations (shows API potential)
|
||||||
|
template <class T> void bench_simdjson_static_reflection_reuse(T &data) {
|
||||||
simdjson::builder::string_builder sb;
|
simdjson::builder::string_builder sb;
|
||||||
simdjson::builder::append(sb, data);
|
simdjson::builder::append(sb, data);
|
||||||
std::string_view p;
|
std::string_view p;
|
||||||
@@ -59,7 +93,7 @@ template <class T> void bench_simdjson_static_reflection(T &data) {
|
|||||||
printf("# output volume: %zu bytes\n", output_volume);
|
printf("# output volume: %zu bytes\n", output_volume);
|
||||||
|
|
||||||
volatile size_t measured_volume = 0;
|
volatile size_t measured_volume = 0;
|
||||||
pretty_print(sizeof(data), output_volume, "bench_simdjson_static_reflection",
|
pretty_print(sizeof(data), output_volume, "bench_simdjson_reuse_buffer",
|
||||||
bench([&data, &measured_volume, &output_volume, &sb]() {
|
bench([&data, &measured_volume, &output_volume, &sb]() {
|
||||||
sb.clear();
|
sb.clear();
|
||||||
simdjson::builder::append(sb, data);
|
simdjson::builder::append(sb, data);
|
||||||
@@ -74,6 +108,63 @@ template <class T> void bench_simdjson_static_reflection(T &data) {
|
|||||||
}));
|
}));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#if SIMDJSON_STATIC_REFLECTION
|
||||||
|
// Fair allocation variant: allocates fresh string each iteration
|
||||||
|
template <class T> void bench_simdjson_to(T &data) {
|
||||||
|
// First run to determine size
|
||||||
|
std::string output_init;
|
||||||
|
if (simdjson::error_code err = simdjson::builder::to_json(data, output_init); err) {
|
||||||
|
std::cerr << "Error in to_json initialization!" << simdjson::error_message(err) << std::endl;
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
size_t output_volume = output_init.size();
|
||||||
|
printf("# output volume: %zu bytes\n", output_volume);
|
||||||
|
|
||||||
|
volatile size_t measured_volume = 0;
|
||||||
|
pretty_print(sizeof(data), output_volume, "bench_simdjson_to",
|
||||||
|
bench([&data, &measured_volume, &output_volume]() {
|
||||||
|
// Fresh allocation each iteration - fair comparison
|
||||||
|
std::string output;
|
||||||
|
if (simdjson::error_code err = simdjson::builder::to_json(data, output); err) {
|
||||||
|
std::cerr << "Error in to_json!" << simdjson::error_message(err) << std::endl;
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
measured_volume = output.size();
|
||||||
|
if (measured_volume != output_volume) {
|
||||||
|
printf("mismatch\n");
|
||||||
|
}
|
||||||
|
}));
|
||||||
|
}
|
||||||
|
|
||||||
|
// Optimized variant: reuses pre-allocated string
|
||||||
|
template <class T> void bench_simdjson_to_reuse(T &data) {
|
||||||
|
std::string output;
|
||||||
|
if (simdjson::error_code err = simdjson::builder::to_json(data, output); err) {
|
||||||
|
std::cerr << "Error in to_json initialization!" << simdjson::error_message(err) << std::endl;
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
size_t output_volume = output.size();
|
||||||
|
printf("# output volume: %zu bytes\n", output_volume);
|
||||||
|
|
||||||
|
// Pre-allocate string with sufficient capacity to avoid reallocation
|
||||||
|
output.reserve(output_volume * 2);
|
||||||
|
|
||||||
|
volatile size_t measured_volume = 0;
|
||||||
|
pretty_print(sizeof(data), output_volume, "bench_simdjson_to_reuse",
|
||||||
|
bench([&data, &measured_volume, &output_volume, &output]() {
|
||||||
|
// Reuse the pre-allocated string - avoids allocation
|
||||||
|
if (simdjson::error_code err = simdjson::builder::to_json(data, output); err) {
|
||||||
|
std::cerr << "Error in to_json!" << simdjson::error_message(err) << std::endl;
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
measured_volume = output.size();
|
||||||
|
if (measured_volume != output_volume) {
|
||||||
|
printf("mismatch\n");
|
||||||
|
}
|
||||||
|
}));
|
||||||
|
}
|
||||||
|
#endif
|
||||||
|
|
||||||
void bench_nlohmann(TwitterData &data) {
|
void bench_nlohmann(TwitterData &data) {
|
||||||
std::string output = nlohmann_serialize(data);
|
std::string output = nlohmann_serialize(data);
|
||||||
size_t output_volume = output.size();
|
size_t output_volume = output.size();
|
||||||
@@ -90,28 +181,61 @@ void bench_nlohmann(TwitterData &data) {
|
|||||||
}));
|
}));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#ifdef SIMDJSON_COMPETITION_YYJSON
|
||||||
|
void bench_yyjson(TwitterData &data) {
|
||||||
|
std::string output = yyjson_serialize(data);
|
||||||
|
size_t output_volume = output.size();
|
||||||
|
printf("# output volume: %zu bytes\n", output_volume);
|
||||||
|
|
||||||
|
volatile size_t measured_volume = 0;
|
||||||
|
pretty_print(1, output_volume, "bench_yyjson",
|
||||||
|
bench([&data, &measured_volume, &output_volume]() {
|
||||||
|
std::string output = yyjson_serialize(data);
|
||||||
|
measured_volume = output.size();
|
||||||
|
if (measured_volume != output_volume) {
|
||||||
|
printf("mismatch\n");
|
||||||
|
}
|
||||||
|
}));
|
||||||
|
}
|
||||||
|
#endif
|
||||||
|
|
||||||
size_t WriteCallback(void *contents, size_t size, size_t nmemb, void *userp) {
|
size_t WriteCallback(void *contents, size_t size, size_t nmemb, void *userp) {
|
||||||
((std::string *)userp)->append((char *)contents, size * nmemb);
|
((std::string *)userp)->append((char *)contents, size * nmemb);
|
||||||
return size * nmemb;
|
return size * nmemb;
|
||||||
}
|
}
|
||||||
|
|
||||||
std::string read_file(std::string filename) {
|
simdjson::padded_string read_file(std::string filename) {
|
||||||
printf("# Reading file %s\n", filename.c_str());
|
printf("# Reading file %s\n", filename.c_str());
|
||||||
constexpr size_t read_size = 4096;
|
constexpr size_t read_size = 4096;
|
||||||
auto stream = std::ifstream(filename.c_str());
|
auto stream = std::ifstream(filename.c_str());
|
||||||
stream.exceptions(std::ios_base::badbit);
|
stream.exceptions(std::ios_base::badbit);
|
||||||
std::string out;
|
simdjson::padded_string_builder builder;
|
||||||
std::string buf(read_size, '\0');
|
std::string buf(read_size, '\0');
|
||||||
while (stream.read(&buf[0], read_size)) {
|
while (stream.read(&buf[0], read_size)) {
|
||||||
out.append(buf, 0, size_t(stream.gcount()));
|
builder.append(buf.data(), size_t(stream.gcount()));
|
||||||
}
|
}
|
||||||
out.append(buf, 0, size_t(stream.gcount()));
|
builder.append(buf.data(), size_t(stream.gcount()));
|
||||||
return out;
|
return builder.convert();
|
||||||
}
|
}
|
||||||
|
|
||||||
// Function to check if benchmark name contains filter substring
|
// Function to check if benchmark name matches any of the comma-separated filters
|
||||||
bool matches_filter(const std::string& benchmark_name, const std::string& filter) {
|
bool matches_filter(const std::string& benchmark_name, const std::string& filter) {
|
||||||
return filter.empty() || benchmark_name.find(filter) != std::string::npos;
|
if (filter.empty()) return true;
|
||||||
|
|
||||||
|
// Split filter by comma
|
||||||
|
size_t start = 0;
|
||||||
|
size_t end = filter.find(',');
|
||||||
|
while (end != std::string::npos) {
|
||||||
|
std::string token = filter.substr(start, end - start);
|
||||||
|
if (benchmark_name.find(token) != std::string::npos) {
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
start = end + 1;
|
||||||
|
end = filter.find(',', start);
|
||||||
|
}
|
||||||
|
// Check last token
|
||||||
|
std::string token = filter.substr(start);
|
||||||
|
return benchmark_name.find(token) != std::string::npos;
|
||||||
}
|
}
|
||||||
|
|
||||||
int main(int argc, char* argv[]) {
|
int main(int argc, char* argv[]) {
|
||||||
@@ -129,12 +253,12 @@ int main(int argc, char* argv[]) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
// Testing correctness of round-trip (serialization + deserialization)
|
// Testing correctness of round-trip (serialization + deserialization)
|
||||||
std::string json_str = read_file(JSON_FILE);
|
simdjson::padded_string json_str = read_file(JSON_FILE);
|
||||||
|
|
||||||
// Loading up the data into a structure.
|
// Loading up the data into a structure.
|
||||||
simdjson::ondemand::parser parser;
|
simdjson::ondemand::parser parser;
|
||||||
simdjson::ondemand::document doc;
|
simdjson::ondemand::document doc;
|
||||||
if(parser.iterate(simdjson::pad(json_str)).get(doc)) {
|
if(parser.iterate(json_str).get(doc)) {
|
||||||
std::cerr << "Error loading the document!" << std::endl;
|
std::cerr << "Error loading the document!" << std::endl;
|
||||||
return EXIT_FAILURE;
|
return EXIT_FAILURE;
|
||||||
}
|
}
|
||||||
@@ -145,18 +269,41 @@ int main(int argc, char* argv[]) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// Benchmarking the serialization
|
// Benchmarking the serialization
|
||||||
|
// Note: simdjson benchmarks include both "fair" (fresh allocation) and "reuse" (buffer reuse) variants
|
||||||
|
// The "fair" variants allocate fresh memory each iteration, matching other libraries' behavior
|
||||||
|
// The "reuse" variants demonstrate the API's potential when buffer reuse is possible
|
||||||
|
|
||||||
if (matches_filter("nlohmann", filter)) {
|
if (matches_filter("nlohmann", filter)) {
|
||||||
bench_nlohmann(my_struct);
|
bench_nlohmann(my_struct);
|
||||||
}
|
}
|
||||||
|
#ifdef SIMDJSON_COMPETITION_YYJSON
|
||||||
|
if (matches_filter("yyjson", filter)) {
|
||||||
|
bench_yyjson(my_struct);
|
||||||
|
}
|
||||||
|
#endif
|
||||||
if (matches_filter("simdjson_static_reflection", filter)) {
|
if (matches_filter("simdjson_static_reflection", filter)) {
|
||||||
bench_simdjson_static_reflection(my_struct);
|
bench_simdjson_static_reflection(my_struct);
|
||||||
}
|
}
|
||||||
|
if (matches_filter("simdjson_reuse", filter)) {
|
||||||
|
bench_simdjson_static_reflection_reuse(my_struct);
|
||||||
|
}
|
||||||
|
#if SIMDJSON_STATIC_REFLECTION
|
||||||
|
if (matches_filter("simdjson_to", filter)) {
|
||||||
|
bench_simdjson_to(my_struct);
|
||||||
|
}
|
||||||
|
if (matches_filter("simdjson_to_reuse", filter)) {
|
||||||
|
bench_simdjson_to_reuse(my_struct);
|
||||||
|
}
|
||||||
|
#endif
|
||||||
#ifdef SIMDJSON_RUST_VERSION
|
#ifdef SIMDJSON_RUST_VERSION
|
||||||
if (matches_filter("rust", filter)) {
|
if (matches_filter("rust", filter)) {
|
||||||
printf("# WARNING: The Rust benchmark may not be directly comparable since it does not use an equivalent data structure.\n");
|
serde_benchmark::TwitterData * td = serde_benchmark::twitter_from_str(json_str.data(), json_str.size());
|
||||||
serde_benchmark::TwitterData * td = serde_benchmark::twitter_from_str(json_str.c_str(), json_str.size());
|
if (td == nullptr) {
|
||||||
bench_rust(td);
|
printf("# Failed to parse Twitter data for Rust benchmark\n");
|
||||||
serde_benchmark::free_twitter(td);
|
} else {
|
||||||
|
bench_rust(td);
|
||||||
|
serde_benchmark::free_twitter(td);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
#endif
|
#endif
|
||||||
#if SIMDJSON_BENCH_CPP_REFLECT
|
#if SIMDJSON_BENCH_CPP_REFLECT
|
||||||
|
|||||||
@@ -4,6 +4,7 @@
|
|||||||
#include "twitter_data.h"
|
#include "twitter_data.h"
|
||||||
#include <nlohmann/json.hpp>
|
#include <nlohmann/json.hpp>
|
||||||
|
|
||||||
|
// Serialization functions for nlohmann
|
||||||
void to_json(nlohmann::json &j, const User &u) {
|
void to_json(nlohmann::json &j, const User &u) {
|
||||||
j = nlohmann::json{{"id", u.id},
|
j = nlohmann::json{{"id", u.id},
|
||||||
{"name", u.name},
|
{"name", u.name},
|
||||||
@@ -16,101 +17,54 @@ void to_json(nlohmann::json &j, const User &u) {
|
|||||||
{"statuses_count", u.statuses_count}};
|
{"statuses_count", u.statuses_count}};
|
||||||
}
|
}
|
||||||
|
|
||||||
void to_json(nlohmann::json &j, const Hashtag &h) {
|
|
||||||
j = nlohmann::json{{"text", h.text},
|
|
||||||
{"indices_start", h.indices_start},
|
|
||||||
{"indices_end", h.indices_end}};
|
|
||||||
}
|
|
||||||
|
|
||||||
void to_json(nlohmann::json &j, const Url &u) {
|
|
||||||
j = nlohmann::json{{"url", u.url},
|
|
||||||
{"expanded_url", u.expanded_url},
|
|
||||||
{"display_url", u.display_url},
|
|
||||||
{"indices_start", u.indices_start},
|
|
||||||
{"indices_end", u.indices_end}};
|
|
||||||
}
|
|
||||||
|
|
||||||
void to_json(nlohmann::json &j, const UserMention &um) {
|
|
||||||
j = nlohmann::json{{"id", um.id},
|
|
||||||
{"name", um.name},
|
|
||||||
{"screen_name", um.screen_name},
|
|
||||||
{"indices_start", um.indices_start},
|
|
||||||
{"indices_end", um.indices_end}};
|
|
||||||
}
|
|
||||||
|
|
||||||
void to_json(nlohmann::json &j, const Entities &e) {
|
|
||||||
j = nlohmann::json{{"hashtags", e.hashtags},
|
|
||||||
{"urls", e.urls},
|
|
||||||
{"user_mentions", e.user_mentions}};
|
|
||||||
}
|
|
||||||
|
|
||||||
void to_json(nlohmann::json &j, const Status &s) {
|
void to_json(nlohmann::json &j, const Status &s) {
|
||||||
j = nlohmann::json{{"created_at", s.created_at},
|
j = nlohmann::json{{"created_at", s.created_at},
|
||||||
{"id", s.id},
|
{"id", s.id},
|
||||||
{"text", s.text},
|
{"text", s.text},
|
||||||
{"user", s.user},
|
{"user", s.user},
|
||||||
{"entities", s.entities},
|
|
||||||
{"retweet_count", s.retweet_count},
|
{"retweet_count", s.retweet_count},
|
||||||
{"favorite_count", s.favorite_count},
|
{"favorite_count", s.favorite_count}};
|
||||||
{"favorited", s.favorited},
|
|
||||||
{"retweeted", s.retweeted}};
|
|
||||||
}
|
|
||||||
|
|
||||||
|
|
||||||
std::string nlohmann_serialize(const std::vector<Hashtag>& v) {
|
|
||||||
nlohmann::json a = nlohmann::json::array();
|
|
||||||
for(const Hashtag & h : v) {
|
|
||||||
a.push_back(nlohmann::json{{"text", h.text},
|
|
||||||
{"indices_start", h.indices_start},
|
|
||||||
{"indices_end", h.indices_end}});
|
|
||||||
}
|
|
||||||
return a.dump();
|
|
||||||
}
|
|
||||||
std::string nlohmann_serialize(const std::vector<Url>& v) {
|
|
||||||
nlohmann::json a = nlohmann::json::array();
|
|
||||||
for(const Url & u : v) {
|
|
||||||
a.push_back(nlohmann::json{{"url", u.url},
|
|
||||||
{"expanded_url", u.expanded_url},
|
|
||||||
{"display_url", u.display_url},
|
|
||||||
{"indices_start", u.indices_start},
|
|
||||||
{"indices_end", u.indices_end}});
|
|
||||||
}
|
|
||||||
return a.dump();
|
|
||||||
}
|
|
||||||
std::string nlohmann_serialize(const std::vector<UserMention>& v) {
|
|
||||||
nlohmann::json a = nlohmann::json::array();
|
|
||||||
for(const UserMention & um : v) {
|
|
||||||
a.push_back(nlohmann::json{{"id", um.id},
|
|
||||||
{"name", um.name},
|
|
||||||
{"screen_name", um.screen_name},
|
|
||||||
{"indices_start", um.indices_start},
|
|
||||||
{"indices_end", um.indices_end}});
|
|
||||||
}
|
|
||||||
return a.dump();
|
|
||||||
}
|
|
||||||
|
|
||||||
std::string nlohmann_serialize(const std::vector<Status>& v) {
|
|
||||||
nlohmann::json a = nlohmann::json::array();
|
|
||||||
for(const Status & s : v) {
|
|
||||||
a.push_back(nlohmann::json{{"created_at", s.created_at},
|
|
||||||
{"id", s.id},
|
|
||||||
{"text", s.text},
|
|
||||||
{"user", s.user},
|
|
||||||
{"entities", s.entities},
|
|
||||||
{"retweet_count", s.retweet_count},
|
|
||||||
{"favorite_count", s.favorite_count},
|
|
||||||
{"favorited", s.favorited},
|
|
||||||
{"retweeted", s.retweeted}});
|
|
||||||
}
|
|
||||||
return a.dump();
|
|
||||||
}
|
}
|
||||||
|
|
||||||
void to_json(nlohmann::json &j, const TwitterData &t) {
|
void to_json(nlohmann::json &j, const TwitterData &t) {
|
||||||
j = nlohmann::json{{"statuses", t.statuses}};
|
j = nlohmann::json{{"statuses", t.statuses}};
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Deserialization functions for nlohmann
|
||||||
|
void from_json(const nlohmann::json &j, User &u) {
|
||||||
|
j.at("id").get_to(u.id);
|
||||||
|
j.at("name").get_to(u.name);
|
||||||
|
j.at("screen_name").get_to(u.screen_name);
|
||||||
|
j.at("location").get_to(u.location);
|
||||||
|
j.at("description").get_to(u.description);
|
||||||
|
j.at("verified").get_to(u.verified);
|
||||||
|
j.at("followers_count").get_to(u.followers_count);
|
||||||
|
j.at("friends_count").get_to(u.friends_count);
|
||||||
|
j.at("statuses_count").get_to(u.statuses_count);
|
||||||
|
}
|
||||||
|
|
||||||
|
void from_json(const nlohmann::json &j, Status &s) {
|
||||||
|
j.at("created_at").get_to(s.created_at);
|
||||||
|
j.at("id").get_to(s.id);
|
||||||
|
j.at("text").get_to(s.text);
|
||||||
|
j.at("user").get_to(s.user);
|
||||||
|
j.at("retweet_count").get_to(s.retweet_count);
|
||||||
|
j.at("favorite_count").get_to(s.favorite_count);
|
||||||
|
}
|
||||||
|
|
||||||
|
void from_json(const nlohmann::json &j, TwitterData &t) {
|
||||||
|
j.at("statuses").get_to(t.statuses);
|
||||||
|
}
|
||||||
|
|
||||||
|
// Helper functions for benchmarking
|
||||||
std::string nlohmann_serialize(const TwitterData &data) {
|
std::string nlohmann_serialize(const TwitterData &data) {
|
||||||
return nlohmann_serialize(data.statuses);
|
nlohmann::json j = data;
|
||||||
|
return j.dump();
|
||||||
|
}
|
||||||
|
|
||||||
|
TwitterData nlohmann_deserialize(const std::string &json_str) {
|
||||||
|
nlohmann::json j = nlohmann::json::parse(json_str);
|
||||||
|
return j.get<TwitterData>();
|
||||||
}
|
}
|
||||||
|
|
||||||
#endif // NLOHMANN_TWITTER_DATA_H
|
#endif // NLOHMANN_TWITTER_DATA_H
|
||||||
@@ -0,0 +1,142 @@
|
|||||||
|
#ifndef RAPIDJSON_TWITTER_DATA_H
|
||||||
|
#define RAPIDJSON_TWITTER_DATA_H
|
||||||
|
|
||||||
|
#include "twitter_data.h"
|
||||||
|
#include <rapidjson/document.h>
|
||||||
|
#include <rapidjson/writer.h>
|
||||||
|
#include <rapidjson/stringbuffer.h>
|
||||||
|
#include <rapidjson/error/en.h>
|
||||||
|
|
||||||
|
using namespace rapidjson;
|
||||||
|
|
||||||
|
// RapidJSON deserialization for simplified Twitter data
|
||||||
|
TwitterData rapidjson_deserialize(const std::string& json_str) {
|
||||||
|
Document doc;
|
||||||
|
doc.Parse(json_str.c_str());
|
||||||
|
|
||||||
|
if (doc.HasParseError()) {
|
||||||
|
throw std::runtime_error("RapidJSON parse error");
|
||||||
|
}
|
||||||
|
|
||||||
|
TwitterData data;
|
||||||
|
|
||||||
|
if (!doc.HasMember("statuses") || !doc["statuses"].IsArray()) {
|
||||||
|
return data;
|
||||||
|
}
|
||||||
|
|
||||||
|
const Value& statuses = doc["statuses"];
|
||||||
|
data.statuses.reserve(statuses.Size());
|
||||||
|
|
||||||
|
for (SizeType i = 0; i < statuses.Size(); i++) {
|
||||||
|
const Value& status_json = statuses[i];
|
||||||
|
Status status;
|
||||||
|
|
||||||
|
// Parse status fields
|
||||||
|
if (status_json.HasMember("created_at") && status_json["created_at"].IsString())
|
||||||
|
status.created_at = status_json["created_at"].GetString();
|
||||||
|
if (status_json.HasMember("id") && status_json["id"].IsUint64())
|
||||||
|
status.id = status_json["id"].GetUint64();
|
||||||
|
if (status_json.HasMember("text") && status_json["text"].IsString())
|
||||||
|
status.text = status_json["text"].GetString();
|
||||||
|
if (status_json.HasMember("retweet_count") && status_json["retweet_count"].IsUint64())
|
||||||
|
status.retweet_count = status_json["retweet_count"].GetUint64();
|
||||||
|
if (status_json.HasMember("favorite_count") && status_json["favorite_count"].IsUint64())
|
||||||
|
status.favorite_count = status_json["favorite_count"].GetUint64();
|
||||||
|
|
||||||
|
// Parse user
|
||||||
|
if (status_json.HasMember("user") && status_json["user"].IsObject()) {
|
||||||
|
const Value& user_json = status_json["user"];
|
||||||
|
User user;
|
||||||
|
|
||||||
|
if (user_json.HasMember("id") && user_json["id"].IsUint64())
|
||||||
|
user.id = user_json["id"].GetUint64();
|
||||||
|
if (user_json.HasMember("name") && user_json["name"].IsString())
|
||||||
|
user.name = user_json["name"].GetString();
|
||||||
|
if (user_json.HasMember("screen_name") && user_json["screen_name"].IsString())
|
||||||
|
user.screen_name = user_json["screen_name"].GetString();
|
||||||
|
if (user_json.HasMember("location") && user_json["location"].IsString())
|
||||||
|
user.location = user_json["location"].GetString();
|
||||||
|
if (user_json.HasMember("description") && user_json["description"].IsString())
|
||||||
|
user.description = user_json["description"].GetString();
|
||||||
|
if (user_json.HasMember("verified") && user_json["verified"].IsBool())
|
||||||
|
user.verified = user_json["verified"].GetBool();
|
||||||
|
if (user_json.HasMember("followers_count") && user_json["followers_count"].IsUint64())
|
||||||
|
user.followers_count = user_json["followers_count"].GetUint64();
|
||||||
|
if (user_json.HasMember("friends_count") && user_json["friends_count"].IsUint64())
|
||||||
|
user.friends_count = user_json["friends_count"].GetUint64();
|
||||||
|
if (user_json.HasMember("statuses_count") && user_json["statuses_count"].IsUint64())
|
||||||
|
user.statuses_count = user_json["statuses_count"].GetUint64();
|
||||||
|
|
||||||
|
status.user = user;
|
||||||
|
}
|
||||||
|
|
||||||
|
data.statuses.push_back(status);
|
||||||
|
}
|
||||||
|
|
||||||
|
return data;
|
||||||
|
}
|
||||||
|
|
||||||
|
// RapidJSON serialization for simplified Twitter data
|
||||||
|
std::string rapidjson_serialize(const TwitterData& data) {
|
||||||
|
Document doc;
|
||||||
|
doc.SetObject();
|
||||||
|
Document::AllocatorType& allocator = doc.GetAllocator();
|
||||||
|
|
||||||
|
Value statuses_array(kArrayType);
|
||||||
|
|
||||||
|
for (const auto& status : data.statuses) {
|
||||||
|
Value status_obj(kObjectType);
|
||||||
|
|
||||||
|
Value created_at;
|
||||||
|
created_at.SetString(status.created_at.c_str(), allocator);
|
||||||
|
status_obj.AddMember("created_at", created_at, allocator);
|
||||||
|
|
||||||
|
status_obj.AddMember("id", status.id, allocator);
|
||||||
|
|
||||||
|
Value text;
|
||||||
|
text.SetString(status.text.c_str(), allocator);
|
||||||
|
status_obj.AddMember("text", text, allocator);
|
||||||
|
|
||||||
|
// Add user
|
||||||
|
Value user_obj(kObjectType);
|
||||||
|
user_obj.AddMember("id", status.user.id, allocator);
|
||||||
|
|
||||||
|
Value name;
|
||||||
|
name.SetString(status.user.name.c_str(), allocator);
|
||||||
|
user_obj.AddMember("name", name, allocator);
|
||||||
|
|
||||||
|
Value screen_name;
|
||||||
|
screen_name.SetString(status.user.screen_name.c_str(), allocator);
|
||||||
|
user_obj.AddMember("screen_name", screen_name, allocator);
|
||||||
|
|
||||||
|
Value location;
|
||||||
|
location.SetString(status.user.location.c_str(), allocator);
|
||||||
|
user_obj.AddMember("location", location, allocator);
|
||||||
|
|
||||||
|
Value description;
|
||||||
|
description.SetString(status.user.description.c_str(), allocator);
|
||||||
|
user_obj.AddMember("description", description, allocator);
|
||||||
|
|
||||||
|
user_obj.AddMember("verified", status.user.verified, allocator);
|
||||||
|
user_obj.AddMember("followers_count", status.user.followers_count, allocator);
|
||||||
|
user_obj.AddMember("friends_count", status.user.friends_count, allocator);
|
||||||
|
user_obj.AddMember("statuses_count", status.user.statuses_count, allocator);
|
||||||
|
|
||||||
|
status_obj.AddMember("user", user_obj, allocator);
|
||||||
|
|
||||||
|
status_obj.AddMember("retweet_count", status.retweet_count, allocator);
|
||||||
|
status_obj.AddMember("favorite_count", status.favorite_count, allocator);
|
||||||
|
|
||||||
|
statuses_array.PushBack(status_obj, allocator);
|
||||||
|
}
|
||||||
|
|
||||||
|
doc.AddMember("statuses", statuses_array, allocator);
|
||||||
|
|
||||||
|
StringBuffer buffer;
|
||||||
|
Writer<StringBuffer> writer(buffer);
|
||||||
|
doc.Accept(writer);
|
||||||
|
|
||||||
|
return buffer.GetString();
|
||||||
|
}
|
||||||
|
|
||||||
|
#endif // RAPIDJSON_TWITTER_DATA_H
|
||||||
@@ -4,68 +4,31 @@
|
|||||||
#include <string>
|
#include <string>
|
||||||
#include <vector>
|
#include <vector>
|
||||||
|
|
||||||
|
// Simplified Twitter structures for benchmarking
|
||||||
|
|
||||||
struct User {
|
struct User {
|
||||||
int64_t id;
|
uint64_t id;
|
||||||
std::string id_str;
|
|
||||||
std::string name;
|
std::string name;
|
||||||
std::string screen_name;
|
std::string screen_name;
|
||||||
std::string location;
|
std::string location;
|
||||||
std::string description;
|
std::string description;
|
||||||
bool verified;
|
bool verified;
|
||||||
int64_t followers_count;
|
uint64_t followers_count;
|
||||||
int64_t friends_count;
|
uint64_t friends_count;
|
||||||
int64_t statuses_count;
|
uint64_t statuses_count;
|
||||||
bool operator<=>(const User &other) const = default;
|
|
||||||
};
|
|
||||||
|
|
||||||
struct Hashtag {
|
|
||||||
std::string text;
|
|
||||||
int64_t indices_start;
|
|
||||||
int64_t indices_end;
|
|
||||||
bool operator<=>(const Hashtag &other) const = default;
|
|
||||||
};
|
|
||||||
|
|
||||||
struct Url {
|
|
||||||
std::string url;
|
|
||||||
std::string expanded_url;
|
|
||||||
std::string display_url;
|
|
||||||
int64_t indices_start;
|
|
||||||
int64_t indices_end;
|
|
||||||
bool operator<=>(const Url &other) const = default;
|
|
||||||
};
|
|
||||||
|
|
||||||
struct UserMention {
|
|
||||||
int64_t id;
|
|
||||||
std::string name;
|
|
||||||
std::string screen_name;
|
|
||||||
int64_t indices_start;
|
|
||||||
int64_t indices_end;
|
|
||||||
bool operator<=>(const UserMention &other) const = default;
|
|
||||||
};
|
|
||||||
|
|
||||||
struct Entities {
|
|
||||||
std::vector<Hashtag> hashtags;
|
|
||||||
std::vector<Url> urls;
|
|
||||||
std::vector<UserMention> user_mentions;
|
|
||||||
bool operator==(const Entities &other) const = default;
|
|
||||||
};
|
};
|
||||||
|
|
||||||
struct Status {
|
struct Status {
|
||||||
std::string created_at;
|
std::string created_at;
|
||||||
int64_t id;
|
uint64_t id;
|
||||||
std::string text;
|
std::string text;
|
||||||
User user;
|
User user;
|
||||||
Entities entities;
|
uint64_t retweet_count;
|
||||||
int64_t retweet_count;
|
uint64_t favorite_count;
|
||||||
int64_t favorite_count;
|
|
||||||
bool favorited;
|
|
||||||
bool retweeted;
|
|
||||||
bool operator==(const Status &other) const = default;
|
|
||||||
};
|
};
|
||||||
|
|
||||||
struct TwitterData {
|
struct TwitterData {
|
||||||
std::vector<Status> statuses;
|
std::vector<Status> statuses;
|
||||||
bool operator==(const TwitterData &other) const = default;
|
|
||||||
};
|
};
|
||||||
|
|
||||||
#endif
|
#endif // TWITTER_DATA_H
|
||||||
@@ -0,0 +1,145 @@
|
|||||||
|
#ifndef YYJSON_TWITTER_DATA_H
|
||||||
|
#define YYJSON_TWITTER_DATA_H
|
||||||
|
|
||||||
|
#include "twitter_data.h"
|
||||||
|
#include <yyjson.h>
|
||||||
|
#include <string>
|
||||||
|
#include <stdexcept>
|
||||||
|
|
||||||
|
// yyjson deserialization for simplified Twitter data
|
||||||
|
TwitterData yyjson_deserialize(const std::string &json_str) {
|
||||||
|
TwitterData data;
|
||||||
|
|
||||||
|
yyjson_doc *doc = yyjson_read(json_str.c_str(), json_str.size(), 0);
|
||||||
|
if (!doc) {
|
||||||
|
throw std::runtime_error("yyjson parse error");
|
||||||
|
}
|
||||||
|
|
||||||
|
yyjson_val *root = yyjson_doc_get_root(doc);
|
||||||
|
if (!root) {
|
||||||
|
yyjson_doc_free(doc);
|
||||||
|
return data;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Get statuses array
|
||||||
|
yyjson_val *statuses_val = yyjson_obj_get(root, "statuses");
|
||||||
|
if (!statuses_val || !yyjson_is_arr(statuses_val)) {
|
||||||
|
yyjson_doc_free(doc);
|
||||||
|
return data;
|
||||||
|
}
|
||||||
|
|
||||||
|
size_t idx, max;
|
||||||
|
yyjson_val *status_val;
|
||||||
|
yyjson_arr_foreach(statuses_val, idx, max, status_val) {
|
||||||
|
Status status;
|
||||||
|
|
||||||
|
// Parse status fields
|
||||||
|
yyjson_val *val;
|
||||||
|
|
||||||
|
val = yyjson_obj_get(status_val, "created_at");
|
||||||
|
if (val && yyjson_is_str(val)) status.created_at = yyjson_get_str(val);
|
||||||
|
|
||||||
|
val = yyjson_obj_get(status_val, "id");
|
||||||
|
if (val && yyjson_is_uint(val)) status.id = yyjson_get_uint(val);
|
||||||
|
|
||||||
|
val = yyjson_obj_get(status_val, "text");
|
||||||
|
if (val && yyjson_is_str(val)) status.text = yyjson_get_str(val);
|
||||||
|
|
||||||
|
val = yyjson_obj_get(status_val, "retweet_count");
|
||||||
|
if (val && yyjson_is_uint(val)) status.retweet_count = yyjson_get_uint(val);
|
||||||
|
|
||||||
|
val = yyjson_obj_get(status_val, "favorite_count");
|
||||||
|
if (val && yyjson_is_uint(val)) status.favorite_count = yyjson_get_uint(val);
|
||||||
|
|
||||||
|
// Parse user
|
||||||
|
yyjson_val *user_val = yyjson_obj_get(status_val, "user");
|
||||||
|
if (user_val && yyjson_is_obj(user_val)) {
|
||||||
|
User user;
|
||||||
|
|
||||||
|
val = yyjson_obj_get(user_val, "id");
|
||||||
|
if (val && yyjson_is_uint(val)) user.id = yyjson_get_uint(val);
|
||||||
|
|
||||||
|
val = yyjson_obj_get(user_val, "name");
|
||||||
|
if (val && yyjson_is_str(val)) user.name = yyjson_get_str(val);
|
||||||
|
|
||||||
|
val = yyjson_obj_get(user_val, "screen_name");
|
||||||
|
if (val && yyjson_is_str(val)) user.screen_name = yyjson_get_str(val);
|
||||||
|
|
||||||
|
val = yyjson_obj_get(user_val, "location");
|
||||||
|
if (val && yyjson_is_str(val)) user.location = yyjson_get_str(val);
|
||||||
|
|
||||||
|
val = yyjson_obj_get(user_val, "description");
|
||||||
|
if (val && yyjson_is_str(val)) user.description = yyjson_get_str(val);
|
||||||
|
|
||||||
|
val = yyjson_obj_get(user_val, "verified");
|
||||||
|
if (val && yyjson_is_bool(val)) user.verified = yyjson_get_bool(val);
|
||||||
|
|
||||||
|
val = yyjson_obj_get(user_val, "followers_count");
|
||||||
|
if (val && yyjson_is_uint(val)) user.followers_count = yyjson_get_uint(val);
|
||||||
|
|
||||||
|
val = yyjson_obj_get(user_val, "friends_count");
|
||||||
|
if (val && yyjson_is_uint(val)) user.friends_count = yyjson_get_uint(val);
|
||||||
|
|
||||||
|
val = yyjson_obj_get(user_val, "statuses_count");
|
||||||
|
if (val && yyjson_is_uint(val)) user.statuses_count = yyjson_get_uint(val);
|
||||||
|
|
||||||
|
status.user = user;
|
||||||
|
}
|
||||||
|
|
||||||
|
data.statuses.push_back(status);
|
||||||
|
}
|
||||||
|
|
||||||
|
yyjson_doc_free(doc);
|
||||||
|
return data;
|
||||||
|
}
|
||||||
|
|
||||||
|
// yyjson serialization for simplified Twitter data
|
||||||
|
std::string yyjson_serialize(const TwitterData &data) {
|
||||||
|
yyjson_mut_doc *doc = yyjson_mut_doc_new(NULL);
|
||||||
|
yyjson_mut_val *root = yyjson_mut_obj(doc);
|
||||||
|
yyjson_mut_doc_set_root(doc, root);
|
||||||
|
|
||||||
|
// Create statuses array
|
||||||
|
yyjson_mut_val *statuses_array = yyjson_mut_arr(doc);
|
||||||
|
|
||||||
|
for (const auto& status : data.statuses) {
|
||||||
|
yyjson_mut_val *status_obj = yyjson_mut_obj(doc);
|
||||||
|
|
||||||
|
// Add status fields
|
||||||
|
yyjson_mut_obj_add_str(doc, status_obj, "created_at", status.created_at.c_str());
|
||||||
|
yyjson_mut_obj_add_uint(doc, status_obj, "id", status.id);
|
||||||
|
yyjson_mut_obj_add_str(doc, status_obj, "text", status.text.c_str());
|
||||||
|
|
||||||
|
// User object
|
||||||
|
yyjson_mut_val *user_obj = yyjson_mut_obj(doc);
|
||||||
|
yyjson_mut_obj_add_uint(doc, user_obj, "id", status.user.id);
|
||||||
|
yyjson_mut_obj_add_str(doc, user_obj, "name", status.user.name.c_str());
|
||||||
|
yyjson_mut_obj_add_str(doc, user_obj, "screen_name", status.user.screen_name.c_str());
|
||||||
|
yyjson_mut_obj_add_str(doc, user_obj, "location", status.user.location.c_str());
|
||||||
|
yyjson_mut_obj_add_str(doc, user_obj, "description", status.user.description.c_str());
|
||||||
|
yyjson_mut_obj_add_bool(doc, user_obj, "verified", status.user.verified);
|
||||||
|
yyjson_mut_obj_add_uint(doc, user_obj, "followers_count", status.user.followers_count);
|
||||||
|
yyjson_mut_obj_add_uint(doc, user_obj, "friends_count", status.user.friends_count);
|
||||||
|
yyjson_mut_obj_add_uint(doc, user_obj, "statuses_count", status.user.statuses_count);
|
||||||
|
yyjson_mut_obj_add_val(doc, status_obj, "user", user_obj);
|
||||||
|
|
||||||
|
// Other fields
|
||||||
|
yyjson_mut_obj_add_uint(doc, status_obj, "retweet_count", status.retweet_count);
|
||||||
|
yyjson_mut_obj_add_uint(doc, status_obj, "favorite_count", status.favorite_count);
|
||||||
|
|
||||||
|
yyjson_mut_arr_append(statuses_array, status_obj);
|
||||||
|
}
|
||||||
|
|
||||||
|
// Add statuses array to root
|
||||||
|
yyjson_mut_obj_add_val(doc, root, "statuses", statuses_array);
|
||||||
|
|
||||||
|
// Write to string
|
||||||
|
char *json_output = yyjson_mut_write(doc, 0, NULL);
|
||||||
|
std::string result(json_output);
|
||||||
|
free(json_output);
|
||||||
|
yyjson_mut_doc_free(doc);
|
||||||
|
|
||||||
|
return result;
|
||||||
|
}
|
||||||
|
|
||||||
|
#endif // YYJSON_TWITTER_DATA_H
|
||||||
+1
-1
@@ -469,7 +469,7 @@ support for users who avoid exceptions. See [the simdjson error handling documen
|
|||||||
out of the array, you may use an array access (e.g., `array[1]`). You should never reset an array as you are iterating through it. The following is an anti-pattern: `for(auto value: myarray) {myarray.reset()}`.
|
out of the array, you may use an array access (e.g., `array[1]`). You should never reset an array as you are iterating through it. The following is an anti-pattern: `for(auto value: myarray) {myarray.reset()}`.
|
||||||
* **Field Access:** To get the value of the "foo" field in an object, use `object["foo"]`. This will
|
* **Field Access:** To get the value of the "foo" field in an object, use `object["foo"]`. This will
|
||||||
scan through the object looking for the field with the matching string, doing a character-by-character
|
scan through the object looking for the field with the matching string, doing a character-by-character
|
||||||
comparison. It may generate the error `simdjson::NO_SUCH_FIELD` if there is no such key in the object, it may throw an exception (see [Error handling](#error-handling)). For efficiency reason, you should avoid looking up the same field repeatedly: e.g., do
|
comparison. It may generate the error `simdjson::NO_SUCH_FIELD` if there is no such key in the object, it may throw an exception (see [Error handling](#error-handling)). The returned value is only valid so long as you do not access another field: normally, you should therefore grab the value right after accessing a key (i.e., convert it to number, string, object, array...). For efficiency reason, you should avoid looking up the same field repeatedly: e.g., do
|
||||||
not do `object["foo"]` followed by `object["foo"]` with the same `object` instance. Generally, you should not mix and match iterating through an object (`for(auto field : object) {...}`) and key accesses (`object["foo"]`): if you need to iterate through an object after a key access, you need to call `reset()` on the object. Whenever you call `reset()`, you need to keep in mind that though you can iterate over the array repeatedly, values should be consumedonly once (e.g., repeatedly calling `unescaped_key()` on the same key is forbidden). Keep in mind that On-Demand does not buffer or save the result of the parsing: if you repeatedly access `object["foo"]`, then it must repeatedly seek the key and parse the content. The library does not provide a distinct function to check if a key is present, instead we recommend you attempt to access the key: e.g., by doing `ondemand::value val{}; if (!object["foo"].get(val)) {...}`, you have that `val` contains the requested value inside the if clause. It is your responsibility as a user to temporarily keep a reference to the value (`auto v = object["foo"]`), or to consume the content and store it in your own data structures. If you consume an
|
not do `object["foo"]` followed by `object["foo"]` with the same `object` instance. Generally, you should not mix and match iterating through an object (`for(auto field : object) {...}`) and key accesses (`object["foo"]`): if you need to iterate through an object after a key access, you need to call `reset()` on the object. Whenever you call `reset()`, you need to keep in mind that though you can iterate over the array repeatedly, values should be consumedonly once (e.g., repeatedly calling `unescaped_key()` on the same key is forbidden). Keep in mind that On-Demand does not buffer or save the result of the parsing: if you repeatedly access `object["foo"]`, then it must repeatedly seek the key and parse the content. The library does not provide a distinct function to check if a key is present, instead we recommend you attempt to access the key: e.g., by doing `ondemand::value val{}; if (!object["foo"].get(val)) {...}`, you have that `val` contains the requested value inside the if clause. It is your responsibility as a user to temporarily keep a reference to the value (`auto v = object["foo"]`), or to consume the content and store it in your own data structures. If you consume an
|
||||||
object twice: `std::string_view(object["foo"]` followed by `std::string_view(object["foo"]` then your code
|
object twice: `std::string_view(object["foo"]` followed by `std::string_view(object["foo"]` then your code
|
||||||
is in error. Furthermore, you can only consume one field at a time, on the same object. The
|
is in error. Furthermore, you can only consume one field at a time, on the same object. The
|
||||||
|
|||||||
@@ -275,7 +275,7 @@ simdjson_warn_unused simdjson_result<std::string> to_json_string(const Z &z, siz
|
|||||||
}
|
}
|
||||||
|
|
||||||
template <class Z>
|
template <class Z>
|
||||||
simdjson_warn_unused simdjson_error to_json(const Z &z, std::string &s, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
|
simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
|
||||||
string_builder b(initial_capacity);
|
string_builder b(initial_capacity);
|
||||||
append(b, z);
|
append(b, z);
|
||||||
std::string_view view;
|
std::string_view view;
|
||||||
@@ -352,7 +352,7 @@ simdjson_warn_unused simdjson_result<std::string> to_json(const Z &z, size_t ini
|
|||||||
return std::string(s);
|
return std::string(s);
|
||||||
}
|
}
|
||||||
template <class Z>
|
template <class Z>
|
||||||
simdjson_warn_unused simdjson_error to_json(const Z &z, std::string &s, size_t initial_capacity = SIMDJSON_IMPLEMENTATION::builder::string_builder::DEFAULT_INITIAL_CAPACITY) {
|
simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initial_capacity = SIMDJSON_IMPLEMENTATION::builder::string_builder::DEFAULT_INITIAL_CAPACITY) {
|
||||||
SIMDJSON_IMPLEMENTATION::builder::string_builder b(initial_capacity);
|
SIMDJSON_IMPLEMENTATION::builder::string_builder b(initial_capacity);
|
||||||
SIMDJSON_IMPLEMENTATION::builder::append(b, z);
|
SIMDJSON_IMPLEMENTATION::builder::append(b, z);
|
||||||
std::string_view view;
|
std::string_view view;
|
||||||
|
|||||||
@@ -29,9 +29,15 @@
|
|||||||
#endif
|
#endif
|
||||||
#if SIMDJSON_EXPERIMENTAL_HAS_NEON
|
#if SIMDJSON_EXPERIMENTAL_HAS_NEON
|
||||||
#include <arm_neon.h>
|
#include <arm_neon.h>
|
||||||
|
#ifdef _MSC_VER
|
||||||
|
#include <intrin.h>
|
||||||
|
#endif
|
||||||
#endif
|
#endif
|
||||||
#if SIMDJSON_EXPERIMENTAL_HAS_SSE2
|
#if SIMDJSON_EXPERIMENTAL_HAS_SSE2
|
||||||
#include <emmintrin.h>
|
#include <emmintrin.h>
|
||||||
|
#ifdef _MSC_VER
|
||||||
|
#include <intrin.h>
|
||||||
|
#endif
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
namespace simdjson {
|
namespace simdjson {
|
||||||
@@ -139,10 +145,10 @@ simdjson_inline bool fast_needs_escaping(std::string_view view) {
|
|||||||
}
|
}
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
|
// Scalar fallback for finding next quotable character
|
||||||
SIMDJSON_CONSTEXPR_LAMBDA inline size_t
|
SIMDJSON_CONSTEXPR_LAMBDA inline size_t
|
||||||
find_next_json_quotable_character(const std::string_view view,
|
find_next_json_quotable_character_scalar(const std::string_view view,
|
||||||
size_t location) noexcept {
|
size_t location) noexcept {
|
||||||
|
|
||||||
for (auto pos = view.begin() + location; pos != view.end(); ++pos) {
|
for (auto pos = view.begin() + location; pos != view.end(); ++pos) {
|
||||||
if (json_quotable_character[static_cast<uint8_t>(*pos)]) {
|
if (json_quotable_character[static_cast<uint8_t>(*pos)]) {
|
||||||
return pos - view.begin();
|
return pos - view.begin();
|
||||||
@@ -151,6 +157,114 @@ find_next_json_quotable_character(const std::string_view view,
|
|||||||
return size_t(view.size());
|
return size_t(view.size());
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// SIMD-accelerated position finding that directly locates the first quotable
|
||||||
|
// character, combining detection and position extraction in a single pass to
|
||||||
|
// minimize redundant work.
|
||||||
|
#if SIMDJSON_EXPERIMENTAL_HAS_NEON
|
||||||
|
simdjson_inline size_t
|
||||||
|
find_next_json_quotable_character(const std::string_view view,
|
||||||
|
size_t location) noexcept {
|
||||||
|
const size_t len = view.size();
|
||||||
|
const uint8_t *ptr =
|
||||||
|
reinterpret_cast<const uint8_t *>(view.data()) + location;
|
||||||
|
size_t remaining = len - location;
|
||||||
|
|
||||||
|
// SIMD constants for characters requiring escape
|
||||||
|
uint8x16_t v34 = vdupq_n_u8(34); // '"'
|
||||||
|
uint8x16_t v92 = vdupq_n_u8(92); // '\\'
|
||||||
|
uint8x16_t v32 = vdupq_n_u8(32); // control char threshold
|
||||||
|
|
||||||
|
while (remaining >= 16) {
|
||||||
|
uint8x16_t word = vld1q_u8(ptr);
|
||||||
|
|
||||||
|
// Check for quotable characters: '"', '\\', or control chars (< 32)
|
||||||
|
uint8x16_t needs_escape = vceqq_u8(word, v34);
|
||||||
|
needs_escape = vorrq_u8(needs_escape, vceqq_u8(word, v92));
|
||||||
|
needs_escape = vorrq_u8(needs_escape, vcltq_u8(word, v32));
|
||||||
|
|
||||||
|
if (vmaxvq_u32(vreinterpretq_u32_u8(needs_escape)) != 0) {
|
||||||
|
// Found quotable character - extract exact byte position using ctz
|
||||||
|
uint64x2_t as64 = vreinterpretq_u64_u8(needs_escape);
|
||||||
|
uint64_t lo = vgetq_lane_u64(as64, 0);
|
||||||
|
uint64_t hi = vgetq_lane_u64(as64, 1);
|
||||||
|
size_t offset = ptr - reinterpret_cast<const uint8_t *>(view.data());
|
||||||
|
#ifdef _MSC_VER
|
||||||
|
unsigned long trailing_zero = 0;
|
||||||
|
if (lo != 0) {
|
||||||
|
_BitScanForward64(&trailing_zero, lo);
|
||||||
|
return offset + trailing_zero / 8;
|
||||||
|
} else {
|
||||||
|
_BitScanForward64(&trailing_zero, hi);
|
||||||
|
return offset + 8 + trailing_zero / 8;
|
||||||
|
}
|
||||||
|
#else
|
||||||
|
if (lo != 0) {
|
||||||
|
return offset + __builtin_ctzll(lo) / 8;
|
||||||
|
} else {
|
||||||
|
return offset + 8 + __builtin_ctzll(hi) / 8;
|
||||||
|
}
|
||||||
|
#endif
|
||||||
|
}
|
||||||
|
ptr += 16;
|
||||||
|
remaining -= 16;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Scalar fallback for remaining bytes
|
||||||
|
size_t current = len - remaining;
|
||||||
|
return find_next_json_quotable_character_scalar(view, current);
|
||||||
|
}
|
||||||
|
#elif SIMDJSON_EXPERIMENTAL_HAS_SSE2
|
||||||
|
simdjson_inline size_t
|
||||||
|
find_next_json_quotable_character(const std::string_view view,
|
||||||
|
size_t location) noexcept {
|
||||||
|
const size_t len = view.size();
|
||||||
|
const uint8_t *ptr =
|
||||||
|
reinterpret_cast<const uint8_t *>(view.data()) + location;
|
||||||
|
size_t remaining = len - location;
|
||||||
|
|
||||||
|
// SIMD constants
|
||||||
|
__m128i v34 = _mm_set1_epi8(34); // '"'
|
||||||
|
__m128i v92 = _mm_set1_epi8(92); // '\\'
|
||||||
|
__m128i v31 = _mm_set1_epi8(31); // for control char detection
|
||||||
|
|
||||||
|
while (remaining >= 16) {
|
||||||
|
__m128i word = _mm_loadu_si128(reinterpret_cast<const __m128i *>(ptr));
|
||||||
|
|
||||||
|
// Check for quotable characters
|
||||||
|
__m128i needs_escape = _mm_cmpeq_epi8(word, v34);
|
||||||
|
needs_escape = _mm_or_si128(needs_escape, _mm_cmpeq_epi8(word, v92));
|
||||||
|
needs_escape = _mm_or_si128(
|
||||||
|
needs_escape,
|
||||||
|
_mm_cmpeq_epi8(_mm_subs_epu8(word, v31), _mm_setzero_si128()));
|
||||||
|
|
||||||
|
int mask = _mm_movemask_epi8(needs_escape);
|
||||||
|
if (mask != 0) {
|
||||||
|
// Found quotable character - use trailing zero count to find position
|
||||||
|
size_t offset = ptr - reinterpret_cast<const uint8_t *>(view.data());
|
||||||
|
#ifdef _MSC_VER
|
||||||
|
unsigned long trailing_zero = 0;
|
||||||
|
_BitScanForward(&trailing_zero, mask);
|
||||||
|
return offset + trailing_zero;
|
||||||
|
#else
|
||||||
|
return offset + __builtin_ctz(mask);
|
||||||
|
#endif
|
||||||
|
}
|
||||||
|
ptr += 16;
|
||||||
|
remaining -= 16;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Scalar fallback for remaining bytes
|
||||||
|
size_t current = len - remaining;
|
||||||
|
return find_next_json_quotable_character_scalar(view, current);
|
||||||
|
}
|
||||||
|
#else
|
||||||
|
SIMDJSON_CONSTEXPR_LAMBDA inline size_t
|
||||||
|
find_next_json_quotable_character(const std::string_view view,
|
||||||
|
size_t location) noexcept {
|
||||||
|
return find_next_json_quotable_character_scalar(view, location);
|
||||||
|
}
|
||||||
|
#endif
|
||||||
|
|
||||||
SIMDJSON_CONSTEXPR_LAMBDA static std::string_view control_chars[] = {
|
SIMDJSON_CONSTEXPR_LAMBDA static std::string_view control_chars[] = {
|
||||||
"\\u0000", "\\u0001", "\\u0002", "\\u0003", "\\u0004", "\\u0005", "\\u0006",
|
"\\u0000", "\\u0001", "\\u0002", "\\u0003", "\\u0004", "\\u0005", "\\u0006",
|
||||||
"\\u0007", "\\b", "\\t", "\\n", "\\u000b", "\\f", "\\r",
|
"\\u0007", "\\b", "\\t", "\\n", "\\u000b", "\\f", "\\r",
|
||||||
@@ -177,14 +291,20 @@ SIMDJSON_CONSTEXPR_LAMBDA void escape_json_char(char c, char *&out) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Writes the escaped version of input to out, returning the number of bytes
|
||||||
|
// written. Uses SIMD position finding to locate quotable characters efficiently.
|
||||||
inline size_t write_string_escaped(const std::string_view input, char *out) {
|
inline size_t write_string_escaped(const std::string_view input, char *out) {
|
||||||
size_t mysize = input.size();
|
size_t mysize = input.size();
|
||||||
if (!fast_needs_escaping(input)) { // fast path!
|
|
||||||
|
// Use SIMD position finder directly - it returns mysize if no escape needed
|
||||||
|
size_t location = find_next_json_quotable_character(input, 0);
|
||||||
|
if (location == mysize) {
|
||||||
|
// Fast path: no escaping needed
|
||||||
memcpy(out, input.data(), input.size());
|
memcpy(out, input.data(), input.size());
|
||||||
return input.size();
|
return input.size();
|
||||||
}
|
}
|
||||||
|
|
||||||
const char *const initout = out;
|
const char *const initout = out;
|
||||||
size_t location = find_next_json_quotable_character(input, 0);
|
|
||||||
memcpy(out, input.data(), location);
|
memcpy(out, input.data(), location);
|
||||||
out += location;
|
out += location;
|
||||||
escape_json_char(input[location], out);
|
escape_json_char(input[location], out);
|
||||||
@@ -383,12 +503,29 @@ simdjson_inline void string_builder::append(number_type v) noexcept {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
else SIMDJSON_IF_CONSTEXPR(std::is_unsigned<number_type>::value) {
|
else SIMDJSON_IF_CONSTEXPR(std::is_unsigned<number_type>::value) {
|
||||||
|
// Process 4 digits at a time instead of 2, reducing store operations
|
||||||
|
// and divisions by approximately half for large numbers.
|
||||||
constexpr size_t max_number_size = 20;
|
constexpr size_t max_number_size = 20;
|
||||||
if (capacity_check(max_number_size)) {
|
if (capacity_check(max_number_size)) {
|
||||||
using unsigned_type = typename std::make_unsigned<number_type>::type;
|
using unsigned_type = typename std::make_unsigned<number_type>::type;
|
||||||
unsigned_type pv = static_cast<unsigned_type>(v);
|
unsigned_type pv = static_cast<unsigned_type>(v);
|
||||||
size_t dc = internal::digit_count(pv);
|
size_t dc = internal::digit_count(pv);
|
||||||
char *write_pointer = buffer.get() + position + dc - 1;
|
char *write_pointer = buffer.get() + position + dc - 1;
|
||||||
|
|
||||||
|
// Process 4 digits per iteration for large numbers
|
||||||
|
while (pv >= 10000) {
|
||||||
|
unsigned_type q = pv / 10000;
|
||||||
|
unsigned_type r = pv % 10000;
|
||||||
|
unsigned_type r_hi = r / 100; // High 2 digits of remainder
|
||||||
|
unsigned_type r_lo = r % 100; // Low 2 digits of remainder
|
||||||
|
// Write low 2 digits first (rightmost), then high 2 digits
|
||||||
|
memcpy(write_pointer - 1, &internal::decimal_table[r_lo * 2], 2);
|
||||||
|
memcpy(write_pointer - 3, &internal::decimal_table[r_hi * 2], 2);
|
||||||
|
write_pointer -= 4;
|
||||||
|
pv = q;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Handle remaining 1-4 digits with original 2-digit loop
|
||||||
while (pv >= 100) {
|
while (pv >= 100) {
|
||||||
memcpy(write_pointer - 1, &internal::decimal_table[(pv % 100) * 2], 2);
|
memcpy(write_pointer - 1, &internal::decimal_table[(pv % 100) * 2], 2);
|
||||||
write_pointer -= 2;
|
write_pointer -= 2;
|
||||||
@@ -403,6 +540,7 @@ simdjson_inline void string_builder::append(number_type v) noexcept {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
else SIMDJSON_IF_CONSTEXPR(std::is_integral<number_type>::value) {
|
else SIMDJSON_IF_CONSTEXPR(std::is_integral<number_type>::value) {
|
||||||
|
// Same 4-digit batching as unsigned path for signed integers
|
||||||
constexpr size_t max_number_size = 20;
|
constexpr size_t max_number_size = 20;
|
||||||
if (capacity_check(max_number_size)) {
|
if (capacity_check(max_number_size)) {
|
||||||
using unsigned_type = typename std::make_unsigned<number_type>::type;
|
using unsigned_type = typename std::make_unsigned<number_type>::type;
|
||||||
@@ -416,6 +554,20 @@ simdjson_inline void string_builder::append(number_type v) noexcept {
|
|||||||
buffer.get()[position] = '-';
|
buffer.get()[position] = '-';
|
||||||
position += negative ? 1 : 0;
|
position += negative ? 1 : 0;
|
||||||
char *write_pointer = buffer.get() + position + dc - 1;
|
char *write_pointer = buffer.get() + position + dc - 1;
|
||||||
|
|
||||||
|
// Process 4 digits per iteration for large numbers
|
||||||
|
while (pv >= 10000) {
|
||||||
|
unsigned_type q = pv / 10000;
|
||||||
|
unsigned_type r = pv % 10000;
|
||||||
|
unsigned_type r_hi = r / 100;
|
||||||
|
unsigned_type r_lo = r % 100;
|
||||||
|
memcpy(write_pointer - 1, &internal::decimal_table[r_lo * 2], 2);
|
||||||
|
memcpy(write_pointer - 3, &internal::decimal_table[r_hi * 2], 2);
|
||||||
|
write_pointer -= 4;
|
||||||
|
pv = q;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Handle remaining 1-4 digits
|
||||||
while (pv >= 100) {
|
while (pv >= 100) {
|
||||||
memcpy(write_pointer - 1, &internal::decimal_table[(pv % 100) * 2], 2);
|
memcpy(write_pointer - 1, &internal::decimal_table[(pv % 100) * 2], 2);
|
||||||
write_pointer -= 2;
|
write_pointer -= 2;
|
||||||
|
|||||||
@@ -280,7 +280,7 @@ simdjson_warn_unused simdjson_result<std::string> to_json(const Z &z, size_t ini
|
|||||||
return std::string(s);
|
return std::string(s);
|
||||||
}
|
}
|
||||||
template <class Z>
|
template <class Z>
|
||||||
simdjson_warn_unused simdjson_error to_json(const Z &z, std::string &s, size_t initial_capacity = simdjson::SIMDJSON_IMPLEMENTATION::builder::string_builder::DEFAULT_INITIAL_CAPACITY) {
|
simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initial_capacity = simdjson::SIMDJSON_IMPLEMENTATION::builder::string_builder::DEFAULT_INITIAL_CAPACITY) {
|
||||||
simdjson::SIMDJSON_IMPLEMENTATION::builder::string_builder b(initial_capacity);
|
simdjson::SIMDJSON_IMPLEMENTATION::builder::string_builder b(initial_capacity);
|
||||||
b.append(z);
|
b.append(z);
|
||||||
std::string_view sv;
|
std::string_view sv;
|
||||||
|
|||||||
@@ -266,6 +266,13 @@ constexpr bool user_defined_type = (std::is_class_v<T>
|
|||||||
&& !std::is_same_v<T, std::string> && !std::is_same_v<T, std::string_view> && !concepts::optional_type<T> &&
|
&& !std::is_same_v<T, std::string> && !std::is_same_v<T, std::string_view> && !concepts::optional_type<T> &&
|
||||||
!concepts::appendable_containers<T>);
|
!concepts::appendable_containers<T>);
|
||||||
|
|
||||||
|
// Trait to indicate JSON fields arrive in struct declaration order.
|
||||||
|
// Users can specialize this for their types to enable faster ordered field lookup.
|
||||||
|
// Example:
|
||||||
|
// template<> struct simdjson::fields_in_order<MyStruct> : std::true_type {};
|
||||||
|
template <typename T>
|
||||||
|
struct fields_in_order : std::false_type {};
|
||||||
|
|
||||||
|
|
||||||
template <typename T, typename ValT>
|
template <typename T, typename ValT>
|
||||||
requires(user_defined_type<T> && std::is_class_v<T>)
|
requires(user_defined_type<T> && std::is_class_v<T>)
|
||||||
@@ -276,25 +283,112 @@ error_code tag_invoke(deserialize_tag, ValT &val, T &out) noexcept {
|
|||||||
} else {
|
} else {
|
||||||
SIMDJSON_TRY(val.get_object().get(obj));
|
SIMDJSON_TRY(val.get_object().get(obj));
|
||||||
}
|
}
|
||||||
template for (constexpr auto mem : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
|
|
||||||
if constexpr (!std::meta::is_const(mem) && std::meta::is_public(mem)) {
|
// For ordered fields, use fast ordered lookup
|
||||||
constexpr std::string_view key = std::define_static_string(std::meta::identifier_of(mem));
|
if constexpr (fields_in_order<T>::value) {
|
||||||
if constexpr (concepts::optional_type<decltype(out.[:mem:])>) {
|
template for (constexpr auto mem : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
|
||||||
// for optional members, it's ok if the key is missing
|
if constexpr (!std::meta::is_const(mem) && std::meta::is_public(mem)) {
|
||||||
auto error = obj[key].get(out.[:mem:]);
|
constexpr std::string_view key = std::define_static_string(std::meta::identifier_of(mem));
|
||||||
if (error && error != NO_SUCH_FIELD) {
|
if constexpr (concepts::optional_type<decltype(out.[:mem:])>) {
|
||||||
if(error == NO_SUCH_FIELD) {
|
error_code error = obj.find_field(key).get(out.[:mem:]);
|
||||||
out.[:mem:].reset();
|
if (error && error != NO_SUCH_FIELD) {
|
||||||
continue;
|
return error;
|
||||||
}
|
}
|
||||||
return error;
|
} else {
|
||||||
|
SIMDJSON_TRY(obj.find_field(key).get(out.[:mem:]));
|
||||||
}
|
}
|
||||||
} else {
|
|
||||||
// for non-optional members, the key must be present
|
|
||||||
SIMDJSON_TRY(obj[key].get(out.[:mem:]));
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
};
|
return simdjson::SUCCESS;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Algorithm selection based on struct size:
|
||||||
|
// - Per-field lookup: calls find_field_unordered() for each struct field (N scans)
|
||||||
|
// - Single-pass: iterates JSON once, checking each field against all struct fields
|
||||||
|
//
|
||||||
|
// Crossover analysis: per-field does N object scans, single-pass does 1 scan with N compares.
|
||||||
|
// Empirically, per-field wins for N<=8, single-pass wins dramatically for N>=9.
|
||||||
|
// At N=9, per-field causes ~35% performance regression vs single-pass.
|
||||||
|
constexpr size_t num_fields = []() consteval {
|
||||||
|
size_t count = 0;
|
||||||
|
template for (constexpr auto mem : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
|
||||||
|
if constexpr (!std::meta::is_const(mem) && std::meta::is_public(mem)) {
|
||||||
|
++count;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return count;
|
||||||
|
}();
|
||||||
|
|
||||||
|
if constexpr (num_fields <= 8) {
|
||||||
|
// Per-field lookup: efficient for small structs, leverages simdjson's SIMD-optimized find_field
|
||||||
|
template for (constexpr auto mem : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
|
||||||
|
if constexpr (!std::meta::is_const(mem) && std::meta::is_public(mem)) {
|
||||||
|
constexpr std::string_view key = std::define_static_string(std::meta::identifier_of(mem));
|
||||||
|
if constexpr (concepts::optional_type<decltype(out.[:mem:])>) {
|
||||||
|
error_code error = obj.find_field_unordered(key).get(out.[:mem:]);
|
||||||
|
if (error && error != NO_SUCH_FIELD) {
|
||||||
|
return error;
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
SIMDJSON_TRY(obj.find_field_unordered(key).get(out.[:mem:]));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
// Single-pass with all optimizations for larger structs
|
||||||
|
constexpr uint64_t required_mask = []() consteval {
|
||||||
|
uint64_t mask = 0;
|
||||||
|
size_t idx = 0;
|
||||||
|
template for (constexpr auto mem : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
|
||||||
|
if constexpr (!std::meta::is_const(mem) && std::meta::is_public(mem)) {
|
||||||
|
if constexpr (!concepts::optional_type<decltype(std::declval<T&>().[:mem:])>) {
|
||||||
|
mask |= (uint64_t(1) << idx);
|
||||||
|
}
|
||||||
|
++idx;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return mask;
|
||||||
|
}();
|
||||||
|
|
||||||
|
uint64_t found_mask = 0;
|
||||||
|
constexpr uint64_t all_fields_mask = (num_fields < 64) ? ((uint64_t(1) << num_fields) - 1) : ~uint64_t(0);
|
||||||
|
|
||||||
|
for (auto field : obj) {
|
||||||
|
if (found_mask == all_fields_mask) break;
|
||||||
|
|
||||||
|
SIMDJSON_IMPLEMENTATION::ondemand::raw_json_string json_key = field.key();
|
||||||
|
const char* raw = json_key.raw();
|
||||||
|
char first_char = raw[0];
|
||||||
|
bool matched = false;
|
||||||
|
size_t field_idx = 0;
|
||||||
|
|
||||||
|
template for (constexpr auto mem : std::define_static_array(std::meta::nonstatic_data_members_of(^^T, std::meta::access_context::unchecked()))) {
|
||||||
|
if constexpr (!std::meta::is_const(mem) && std::meta::is_public(mem)) {
|
||||||
|
constexpr std::string_view expected_key = std::define_static_string(std::meta::identifier_of(mem));
|
||||||
|
if (!matched && first_char == expected_key[0] && !(found_mask & (uint64_t(1) << field_idx))) {
|
||||||
|
if (json_key.unsafe_is_equal(expected_key)) {
|
||||||
|
auto field_val = field.value();
|
||||||
|
error_code err = field_val.get(out.[:mem:]);
|
||||||
|
if (err) {
|
||||||
|
if constexpr (concepts::optional_type<decltype(out.[:mem:])>) {
|
||||||
|
if (err != INCORRECT_TYPE) { return err; }
|
||||||
|
} else {
|
||||||
|
return err;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
found_mask |= (uint64_t(1) << field_idx);
|
||||||
|
matched = true;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
++field_idx;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
if ((found_mask & required_mask) != required_mask) {
|
||||||
|
return NO_SUCH_FIELD;
|
||||||
|
}
|
||||||
|
}
|
||||||
return simdjson::SUCCESS;
|
return simdjson::SUCCESS;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -198,17 +198,14 @@ public:
|
|||||||
* simdjson::ondemand::document doc = parser.iterate(json);
|
* simdjson::ondemand::document doc = parser.iterate(json);
|
||||||
* auto view = doc["deviceId"].get_string(true);
|
* auto view = doc["deviceId"].get_string(true);
|
||||||
*
|
*
|
||||||
* @returns An UTF-8 string. The string is stored in the parser when escaping was needed
|
* @returns An UTF-8 string. The string is stored in the parser and will be invalidated the next
|
||||||
* and will be invalidated the next
|
* time it parses a document or when it is destroyed.
|
||||||
* time it parses a document or when it is destroyed. If no escaping was needed,
|
|
||||||
* the string_view points directly into the original JSON buffer.
|
|
||||||
* @returns INCORRECT_TYPE if the JSON value is not a string.
|
* @returns INCORRECT_TYPE if the JSON value is not a string.
|
||||||
*/
|
*/
|
||||||
simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
|
simdjson_inline simdjson_result<std::string_view> get_string(bool allow_replacement = false) noexcept;
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Attempts to fill the provided std::string reference with the parsed value of the current string.
|
* Attempts to fill the provided std::string reference with the parsed value of the current string.
|
||||||
* The data is stored into the provided std::string instance.
|
|
||||||
*
|
*
|
||||||
* The string is guaranteed to be valid UTF-8.
|
* The string is guaranteed to be valid UTF-8.
|
||||||
*
|
*
|
||||||
|
|||||||
@@ -510,38 +510,6 @@ simdjson_warn_unused simdjson_inline simdjson_result<bool> value_iterator::parse
|
|||||||
}
|
}
|
||||||
|
|
||||||
simdjson_warn_unused simdjson_inline simdjson_result<std::string_view> value_iterator::get_string(bool allow_replacement) noexcept {
|
simdjson_warn_unused simdjson_inline simdjson_result<std::string_view> value_iterator::get_string(bool allow_replacement) noexcept {
|
||||||
// Optimization strategy:
|
|
||||||
// We expect that most strings do not have escape characters, and most are short.
|
|
||||||
// So we can quickly check for backslashes and if there are none, we can just return a string_view
|
|
||||||
// into the original JSON buffer. There is no need to copy or unescape.
|
|
||||||
|
|
||||||
// Fast path: check for backslash in the string
|
|
||||||
// It may seem that this function is odd in that it scans the string even if a
|
|
||||||
// backslash is found early. However, we expect that in most strings there will be no backslash,
|
|
||||||
// so we optimize for that case. The compiler knows to expect a full scan and it can optimize for it.
|
|
||||||
auto has_backslash_fast = [](std::string_view s) noexcept {
|
|
||||||
for(const char c : s) {
|
|
||||||
if(c == '\\') {
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return false;
|
|
||||||
};
|
|
||||||
std::string_view string_with_quotes(reinterpret_cast<const char*>(peek_start()), peek_start_length());
|
|
||||||
if(string_with_quotes.front() != '"') {
|
|
||||||
return incorrect_type_error("Not a string");
|
|
||||||
}
|
|
||||||
|
|
||||||
if(!has_backslash_fast(string_with_quotes)) {
|
|
||||||
// Find the ending quote
|
|
||||||
size_t len = string_with_quotes.size();
|
|
||||||
while(string_with_quotes[len - 1] != '"') {
|
|
||||||
len--;
|
|
||||||
}
|
|
||||||
// At this point len is 2 or more
|
|
||||||
return std::string_view(string_with_quotes.data() + 1, len - 2);
|
|
||||||
}
|
|
||||||
// Slow path: we have a backslash, so we need to unescape
|
|
||||||
return get_raw_json_string().unescape(json_iter(), allow_replacement);
|
return get_raw_json_string().unescape(json_iter(), allow_replacement);
|
||||||
}
|
}
|
||||||
template <typename string_type>
|
template <typename string_type>
|
||||||
@@ -549,7 +517,8 @@ simdjson_warn_unused simdjson_inline error_code value_iterator::get_string(strin
|
|||||||
std::string_view content;
|
std::string_view content;
|
||||||
// Save the string buffer location so that we can restore it after get_string
|
// Save the string buffer location so that we can restore it after get_string
|
||||||
auto saved_string_buf_loc = _json_iter->string_buf_loc();
|
auto saved_string_buf_loc = _json_iter->string_buf_loc();
|
||||||
SIMDJSON_TRY(get_string(allow_replacement).get(content));
|
auto err = get_string(allow_replacement).get(content);
|
||||||
|
if (err) { return err; }
|
||||||
receiver = content;
|
receiver = content;
|
||||||
// Restore the string buffer location, effectively discarding any temporary string storage
|
// Restore the string buffer location, effectively discarding any temporary string storage
|
||||||
_json_iter->string_buf_loc() = saved_string_buf_loc;
|
_json_iter->string_buf_loc() = saved_string_buf_loc;
|
||||||
@@ -687,8 +656,6 @@ simdjson_inline simdjson_result<number> value_iterator::get_root_number(bool che
|
|||||||
return num;
|
return num;
|
||||||
}
|
}
|
||||||
simdjson_warn_unused simdjson_inline simdjson_result<std::string_view> value_iterator::get_root_string(bool check_trailing, bool allow_replacement) noexcept {
|
simdjson_warn_unused simdjson_inline simdjson_result<std::string_view> value_iterator::get_root_string(bool check_trailing, bool allow_replacement) noexcept {
|
||||||
// We could optimize for the no-escape case here as well, but root strings
|
|
||||||
// are less common so we do the simple thing for now.
|
|
||||||
return get_root_raw_json_string(check_trailing).unescape(json_iter(), allow_replacement);
|
return get_root_raw_json_string(check_trailing).unescape(json_iter(), allow_replacement);
|
||||||
}
|
}
|
||||||
template <typename string_type>
|
template <typename string_type>
|
||||||
@@ -696,7 +663,8 @@ simdjson_warn_unused simdjson_inline error_code value_iterator::get_root_string(
|
|||||||
std::string_view content;
|
std::string_view content;
|
||||||
// Save the string buffer location so that we can restore it after get_string
|
// Save the string buffer location so that we can restore it after get_string
|
||||||
auto saved_string_buf_loc = _json_iter->string_buf_loc();
|
auto saved_string_buf_loc = _json_iter->string_buf_loc();
|
||||||
SIMDJSON_TRY(get_root_string(check_trailing, allow_replacement).get(content));
|
auto err = get_root_string(check_trailing, allow_replacement).get(content);
|
||||||
|
if (err) { return err; }
|
||||||
receiver = content;
|
receiver = content;
|
||||||
// Restore the string buffer location, effectively discarding any temporary string storage
|
// Restore the string buffer location, effectively discarding any temporary string storage
|
||||||
_json_iter->string_buf_loc() = saved_string_buf_loc;
|
_json_iter->string_buf_loc() = saved_string_buf_loc;
|
||||||
|
|||||||
+11
-4
@@ -54,15 +54,15 @@ Importantly, we build the experimental LLVM compiler based on the current state
|
|||||||
```bash
|
```bash
|
||||||
CXX=clang++ cmake -B buildreflect -D SIMDJSON_STATIC_REFLECTION=ON -DSIMDJSON_DEVELOPER_MODE=ON
|
CXX=clang++ cmake -B buildreflect -D SIMDJSON_STATIC_REFLECTION=ON -DSIMDJSON_DEVELOPER_MODE=ON
|
||||||
```
|
```
|
||||||
This only needs to be done once.
|
This only needs to be done once. To build the Rust code, add `-D SIMDJSON_USE_RUST=ON`. Note that you should have Rust on your system as a prerequisite for this option to be meaningful.
|
||||||
|
|
||||||
5. Build the code...
|
5. Build the code...
|
||||||
```bash
|
```bash
|
||||||
cmake --build buildreflect --target benchmark_serialization_citm_catalog benchmark_serialization_twitter
|
cmake --build buildreflect --target benchmark_serialization_citm_catalog benchmark_serialization_twitter benchmark_parsing_twitter benchmark_parsing_citm
|
||||||
```
|
```
|
||||||
|
This is sufficient if you only mean to run the benchmarks (and skip the tests).
|
||||||
|
|
||||||
|
6. Run the tests... (optional)
|
||||||
6. Run the tests...
|
|
||||||
```bash
|
```bash
|
||||||
cmake --build buildreflect
|
cmake --build buildreflect
|
||||||
ctest --test-dir buildreflect --output-on-failure
|
ctest --test-dir buildreflect --output-on-failure
|
||||||
@@ -71,9 +71,16 @@ ctest --test-dir buildreflect --output-on-failure
|
|||||||
7. Run the benchmarks.
|
7. Run the benchmarks.
|
||||||
```bash
|
```bash
|
||||||
./buildreflect/benchmark/static_reflect/citm_catalog_benchmark/benchmark_serialization_citm_catalog
|
./buildreflect/benchmark/static_reflect/citm_catalog_benchmark/benchmark_serialization_citm_catalog
|
||||||
|
./buildreflect/benchmark/static_reflect/citm_catalog_benchmark/benchmark_parsing_citm
|
||||||
./buildreflect/benchmark/static_reflect/twitter_benchmark/benchmark_serialization_twitter
|
./buildreflect/benchmark/static_reflect/twitter_benchmark/benchmark_serialization_twitter
|
||||||
|
./buildreflect/benchmark/static_reflect/twitter_benchmark/benchmark_parsing_twitter
|
||||||
```
|
```
|
||||||
|
|
||||||
|
These benchmarks should print performance counters *if* run in privileged mode (e.g., run under sudo).
|
||||||
|
If you run the code inside a docker container, under Linux, it should have access to the performance
|
||||||
|
counters if they are enabled on the host. Note that they are usually disabled in a cloud setting
|
||||||
|
unless you have so-called metal access.
|
||||||
|
|
||||||
You can modify the source code with your favorite editor and run again steps 5 (Build the code) and 6 (Run the tests) and 7 (Run the benchmark). Importantly, you should remain in the docker shell.
|
You can modify the source code with your favorite editor and run again steps 5 (Build the code) and 6 (Run the tests) and 7 (Run the benchmark). Importantly, you should remain in the docker shell.
|
||||||
|
|
||||||
You can create a new docker shell at any time by running step 3 (bash script).
|
You can create a new docker shell at any time by running step 3 (bash script).
|
||||||
|
|||||||
+26
-13
@@ -13,6 +13,18 @@ import datetime
|
|||||||
import json
|
import json
|
||||||
from typing import Dict, List, Optional, Set, TextIO, Union, cast
|
from typing import Dict, List, Optional, Set, TextIO, Union, cast
|
||||||
|
|
||||||
|
# Pre-compile regex patterns for performance
|
||||||
|
pragma_once_re = re.compile(r'^#pragma once$')
|
||||||
|
ifndef_conditional_re = re.compile(r'^#ifndef\s+SIMDJSON_CONDITIONAL_INCLUDE\s*$')
|
||||||
|
endif_conditional_re = re.compile(r'^#endif\s*//\s*SIMDJSON_CONDITIONAL_INCLUDE\s*$')
|
||||||
|
include_re = re.compile(r'^#include\s+["<]([^">]*)[">]')
|
||||||
|
define_implementation_re = re.compile(r'^#define\s+SIMDJSON_IMPLEMENTATION\s+(.+)$')
|
||||||
|
undef_implementation_re = re.compile(r'^#undef\s+SIMDJSON_IMPLEMENTATION\s*$')
|
||||||
|
simdjson_implementation_re = re.compile(r'\bSIMDJSON_IMPLEMENTATION\b')
|
||||||
|
define_conditional_re = re.compile(r'^#define\s+SIMDJSON_CONDITIONAL_INCLUDE\s*$')
|
||||||
|
undef_conditional_re = re.compile(r'^#undef\s+SIMDJSON_CONDITIONAL_INCLUDE\s*$')
|
||||||
|
version_re = re.compile(r'\d+\.\d+\.\d+')
|
||||||
|
|
||||||
# Check for Python 3, this does not actually work.
|
# Check for Python 3, this does not actually work.
|
||||||
if sys.version_info < (3, 0):
|
if sys.version_info < (3, 0):
|
||||||
sys.stdout.write("Sorry, requires Python 3.x or better\n")
|
sys.stdout.write("Sorry, requires Python 3.x or better\n")
|
||||||
@@ -109,6 +121,7 @@ RelativeRoot = str # Literal['src','include'] # Literal not supported in Python
|
|||||||
RELATIVE_ROOTS: List[RelativeRoot] = ['src', 'include' ]
|
RELATIVE_ROOTS: List[RelativeRoot] = ['src', 'include' ]
|
||||||
Implementation = str # Literal['arm64', 'fallback', 'haswell', 'icelake', 'ppc64', 'westmere', 'lsx', 'lasx'] # Literal not supported in Python 3.7 (CI)
|
Implementation = str # Literal['arm64', 'fallback', 'haswell', 'icelake', 'ppc64', 'westmere', 'lsx', 'lasx'] # Literal not supported in Python 3.7 (CI)
|
||||||
IMPLEMENTATIONS: List[Implementation] = [ 'arm64', 'haswell', 'icelake', 'lasx', 'lsx', 'ppc64', 'rvv-vls', 'westmere', 'fallback' ]
|
IMPLEMENTATIONS: List[Implementation] = [ 'arm64', 'haswell', 'icelake', 'lasx', 'lsx', 'ppc64', 'rvv-vls', 'westmere', 'fallback' ]
|
||||||
|
implementation_re = re.compile(f'(^|/)({"|".join(IMPLEMENTATIONS)})')
|
||||||
GENERIC_INCLUDE = "simdjson/generic"
|
GENERIC_INCLUDE = "simdjson/generic"
|
||||||
GENERIC_SRC = "generic"
|
GENERIC_SRC = "generic"
|
||||||
BUILTIN = "simdjson/builtin"
|
BUILTIN = "simdjson/builtin"
|
||||||
@@ -199,7 +212,7 @@ class SimdjsonFile:
|
|||||||
|
|
||||||
@property
|
@property
|
||||||
def implementation(self) -> Optional[Implementation]:
|
def implementation(self) -> Optional[Implementation]:
|
||||||
match = re.search(f'(^|/)({"|".join(IMPLEMENTATIONS)})', self.include_path)
|
match = implementation_re.search(self.include_path)
|
||||||
if match:
|
if match:
|
||||||
return cast(Implementation, str(match.group(2)))
|
return cast(Implementation, str(match.group(2)))
|
||||||
|
|
||||||
@@ -417,23 +430,23 @@ class Amalgamator:
|
|||||||
line = line.rstrip('\n')
|
line = line.rstrip('\n')
|
||||||
|
|
||||||
# Ignore #pragma once, it causes warnings if it ends up in a .cpp file
|
# Ignore #pragma once, it causes warnings if it ends up in a .cpp file
|
||||||
if re.search(r'^#pragma once$', line):
|
if pragma_once_re.search(line):
|
||||||
continue
|
continue
|
||||||
|
|
||||||
# Ignore lines inside #ifndef SIMDJSON_CONDITIONAL_INCLUDE
|
# Ignore lines inside #ifndef SIMDJSON_CONDITIONAL_INCLUDE
|
||||||
if re.search(r'^#ifndef\s+SIMDJSON_CONDITIONAL_INCLUDE\s*$', line):
|
if ifndef_conditional_re.search(line):
|
||||||
assert file.is_conditional_include, f"Error: File '{file}' uses '#ifndef SIMDJSON_CONDITIONAL_INCLUDE', but it's not an amalgamated file. Conditional includes are only for amalgamated files. {rules}"
|
assert file.is_conditional_include, f"Error: File '{file}' uses '#ifndef SIMDJSON_CONDITIONAL_INCLUDE', but it's not an amalgamated file. Conditional includes are only for amalgamated files. {rules}"
|
||||||
assert self.in_conditional_include_block, f"Error: File '{file}' uses '#ifndef SIMDJSON_CONDITIONAL_INCLUDE' without a prior '#define SIMDJSON_CONDITIONAL_INCLUDE'. Ensure the define comes first. Stack: {self.include_stack}. {rules}"
|
assert self.in_conditional_include_block, f"Error: File '{file}' uses '#ifndef SIMDJSON_CONDITIONAL_INCLUDE' without a prior '#define SIMDJSON_CONDITIONAL_INCLUDE'. Ensure the define comes first. Stack: {self.include_stack}. {rules}"
|
||||||
assert not self.editor_only_region, f"Error: File '{file}' uses '#ifndef SIMDJSON_CONDITIONAL_INCLUDE' twice in a row. Ensure conditional blocks are properly nested and closed. {rules}"
|
assert not self.editor_only_region, f"Error: File '{file}' uses '#ifndef SIMDJSON_CONDITIONAL_INCLUDE' twice in a row. Ensure conditional blocks are properly nested and closed. {rules}"
|
||||||
self.editor_only_region = True
|
self.editor_only_region = True
|
||||||
|
|
||||||
# Handle ignored lines (and ending ignore blocks)
|
# Handle ignored lines (and ending ignore blocks)
|
||||||
end_ignore = re.search(r'^#endif\s*//\s*SIMDJSON_CONDITIONAL_INCLUDE\s*$', line)
|
end_ignore = endif_conditional_re.search(line)
|
||||||
if self.editor_only_region:
|
if self.editor_only_region:
|
||||||
self.write(f"/* amalgamation skipped (editor-only): {line} */")
|
self.write(f"/* amalgamation skipped (editor-only): {line} */")
|
||||||
|
|
||||||
# Add the editor-only include so we can check dependencies.h for completeness later
|
# Add the editor-only include so we can check dependencies.h for completeness later
|
||||||
included = re.search(r'^#include\s+["<]([^">]*)[">]', line)
|
included = include_re.search(line)
|
||||||
if included:
|
if included:
|
||||||
included_file = self.repository[included.group(1)]
|
included_file = self.repository[included.group(1)]
|
||||||
if included_file:
|
if included_file:
|
||||||
@@ -445,7 +458,7 @@ class Amalgamator:
|
|||||||
assert not end_ignore, f"Error: File '{file}' has '#endif // SIMDJSON_CONDITIONAL_INCLUDE' without a matching '#ifndef'. Ensure proper conditional block structure. {rules}"
|
assert not end_ignore, f"Error: File '{file}' has '#endif // SIMDJSON_CONDITIONAL_INCLUDE' without a matching '#ifndef'. Ensure proper conditional block structure. {rules}"
|
||||||
|
|
||||||
# Handle #include lines
|
# Handle #include lines
|
||||||
included = re.search(r'^#include\s+["<]([^">]*)[">]', line)
|
included = include_re.search(line)
|
||||||
if included:
|
if included:
|
||||||
# we explicitly include simdjson headers, one time each (unless they are generic, in which case multiple times is fine)
|
# we explicitly include simdjson headers, one time each (unless they are generic, in which case multiple times is fine)
|
||||||
included_file = self.repository[included.group(1)]
|
included_file = self.repository[included.group(1)]
|
||||||
@@ -455,7 +468,7 @@ class Amalgamator:
|
|||||||
continue
|
continue
|
||||||
|
|
||||||
# Handle defining and replacing SIMDJSON_IMPLEMENTATION
|
# Handle defining and replacing SIMDJSON_IMPLEMENTATION
|
||||||
defined = re.search(r'^#define\s+SIMDJSON_IMPLEMENTATION\s+(.+)$', line)
|
defined = define_implementation_re.search(line)
|
||||||
if defined:
|
if defined:
|
||||||
old_implementation = self.implementation
|
old_implementation = self.implementation
|
||||||
self.implementation = defined.group(1)
|
self.implementation = defined.group(1)
|
||||||
@@ -463,24 +476,24 @@ class Amalgamator:
|
|||||||
self.write(f'/* defining SIMDJSON_IMPLEMENTATION to "{self.implementation}" */')
|
self.write(f'/* defining SIMDJSON_IMPLEMENTATION to "{self.implementation}" */')
|
||||||
else:
|
else:
|
||||||
self.write(f'/* redefining SIMDJSON_IMPLEMENTATION from "{old_implementation}" to "{self.implementation}" */')
|
self.write(f'/* redefining SIMDJSON_IMPLEMENTATION from "{old_implementation}" to "{self.implementation}" */')
|
||||||
elif re.search(r'^#undef\s+SIMDJSON_IMPLEMENTATION\s*$', line):
|
elif undef_implementation_re.search(line):
|
||||||
# Don't include #undef SIMDJSON_IMPLEMENTATION since we're handling it ourselves
|
# Don't include #undef SIMDJSON_IMPLEMENTATION since we're handling it ourselves
|
||||||
self.write(f'/* undefining SIMDJSON_IMPLEMENTATION from "{self.implementation}" */')
|
self.write(f'/* undefining SIMDJSON_IMPLEMENTATION from "{self.implementation}" */')
|
||||||
self.implementation = None
|
self.implementation = None
|
||||||
elif re.search(r'\bSIMDJSON_IMPLEMENTATION\b', line) and file.include_path != IMPLEMENTATION_DETECTION_H:
|
elif self.implementation and file.include_path != IMPLEMENTATION_DETECTION_H:
|
||||||
# copy the line, with SIMDJSON_IMPLEMENTATION replace to what it is currently defined to
|
# copy the line, with SIMDJSON_IMPLEMENTATION replace to what it is currently defined to
|
||||||
assert self.implementation, f"Error: In '{file}', line '{line}' uses SIMDJSON_IMPLEMENTATION, but it's not defined. Ensure SIMDJSON_IMPLEMENTATION is set before use. {rules}"
|
assert self.implementation, f"Error: In '{file}', line '{line}' uses SIMDJSON_IMPLEMENTATION, but it's not defined. Ensure SIMDJSON_IMPLEMENTATION is set before use. {rules}"
|
||||||
line = re.sub(r'\bSIMDJSON_IMPLEMENTATION\b',self.implementation,line)
|
line = simdjson_implementation_re.sub(self.implementation, line)
|
||||||
|
|
||||||
# Handle defining and undefining SIMDJSON_CONDITIONAL_INCLUDE
|
# Handle defining and undefining SIMDJSON_CONDITIONAL_INCLUDE
|
||||||
defined = re.search(r'^#define\s+SIMDJSON_CONDITIONAL_INCLUDE\s*$', line)
|
defined = define_conditional_re.search(line)
|
||||||
if defined:
|
if defined:
|
||||||
assert not file.is_conditional_include, f"Error: Amalgamated file '{file}' defines SIMDJSON_CONDITIONAL_INCLUDE, which is not allowed. Only non-amalgamated files can define it. {rules}"
|
assert not file.is_conditional_include, f"Error: Amalgamated file '{file}' defines SIMDJSON_CONDITIONAL_INCLUDE, which is not allowed. Only non-amalgamated files can define it. {rules}"
|
||||||
assert not self.in_conditional_include_block, f"Error: File '{file}' redefines SIMDJSON_CONDITIONAL_INCLUDE while already in a conditional block. Avoid redefinition. {rules}"
|
assert not self.in_conditional_include_block, f"Error: File '{file}' redefines SIMDJSON_CONDITIONAL_INCLUDE while already in a conditional block. Avoid redefinition. {rules}"
|
||||||
self.in_conditional_include_block = True
|
self.in_conditional_include_block = True
|
||||||
self.found_includes_per_conditional_block.clear()
|
self.found_includes_per_conditional_block.clear()
|
||||||
self.write(f'/* defining SIMDJSON_CONDITIONAL_INCLUDE */')
|
self.write(f'/* defining SIMDJSON_CONDITIONAL_INCLUDE */')
|
||||||
elif re.search(r'^#undef\s+SIMDJSON_CONDITIONAL_INCLUDE\s*$', line):
|
elif undef_conditional_re.search(line):
|
||||||
assert not file.is_conditional_include, f"Error: Amalgamated file '{file}' undefines SIMDJSON_CONDITIONAL_INCLUDE, which is not allowed. Only non-amalgamated files can undefine it. {rules}"
|
assert not file.is_conditional_include, f"Error: Amalgamated file '{file}' undefines SIMDJSON_CONDITIONAL_INCLUDE, which is not allowed. Only non-amalgamated files can undefine it. {rules}"
|
||||||
assert self.in_conditional_include_block, f"Error: File '{file}' undefines SIMDJSON_CONDITIONAL_INCLUDE without having defined it first. Ensure proper define/undefine pairing. {rules}"
|
assert self.in_conditional_include_block, f"Error: File '{file}' undefines SIMDJSON_CONDITIONAL_INCLUDE without having defined it first. Ensure proper define/undefine pairing. {rules}"
|
||||||
self.write(f'/* undefining SIMDJSON_CONDITIONAL_INCLUDE */')
|
self.write(f'/* undefining SIMDJSON_CONDITIONAL_INCLUDE */')
|
||||||
@@ -554,7 +567,7 @@ def validate_implementations():
|
|||||||
|
|
||||||
def read_version():
|
def read_version():
|
||||||
with open(os.path.join(PROJECTPATH, 'include/simdjson/simdjson_version.h')) as f:
|
with open(os.path.join(PROJECTPATH, 'include/simdjson/simdjson_version.h')) as f:
|
||||||
return re.search(r'\d+\.\d+\.\d+', f.read()).group(0)
|
return version_re.search(f.read()).group(0)
|
||||||
|
|
||||||
version = read_version()
|
version = read_version()
|
||||||
if not validate_implementations():
|
if not validate_implementations():
|
||||||
|
|||||||
@@ -145,7 +145,7 @@ namespace internal {
|
|||||||
+ SIMDJSON_IMPLEMENTATION_RVV_VLS + SIMDJSON_IMPLEMENTATION_FALLBACK == 1)
|
+ SIMDJSON_IMPLEMENTATION_RVV_VLS + SIMDJSON_IMPLEMENTATION_FALLBACK == 1)
|
||||||
|
|
||||||
#if SIMDJSON_SINGLE_IMPLEMENTATION
|
#if SIMDJSON_SINGLE_IMPLEMENTATION
|
||||||
static const implementation* get_single_implementation() {
|
simdjson_really_inline static const implementation* get_single_implementation() {
|
||||||
return
|
return
|
||||||
#if SIMDJSON_IMPLEMENTATION_ICELAKE
|
#if SIMDJSON_IMPLEMENTATION_ICELAKE
|
||||||
get_icelake_singleton();
|
get_icelake_singleton();
|
||||||
|
|||||||
Reference in New Issue
Block a user