mirror of
https://github.com/simdjson/simdjson
synced 2026-06-08 17:27:07 +00:00
Compare commits
14 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 06c774c655 | |||
| 95d8c81810 | |||
| af2a43610e | |||
| e362b18499 | |||
| 6e60ed8338 | |||
| e2ca00c841 | |||
| b7124f7c64 | |||
| e6c6db6493 | |||
| 6fffc99dac | |||
| 8c5cc8c443 | |||
| d8c49a8f25 | |||
| cda98064b8 | |||
| e532d61e16 | |||
| db93de2a21 |
@@ -19,11 +19,11 @@ jobs:
|
||||
sudo apt-get install -y cmake make g++-riscv64-linux-gnu qemu-user-static clang-18
|
||||
- name: Build
|
||||
run: |
|
||||
CXX=clang++-18 CXXFLAGS="--target=riscv64-linux-gnu -march=rv64gcv_zvbb" \
|
||||
cmake --toolchain=cmake/toolchains-ci/riscv64-linux-gnu.cmake -DCMAKE_BUILD_TYPE=Release -B build
|
||||
cmake --build build/ -j$(nproc)
|
||||
CC=clang-18 CXX=clang++-18 CFLAGS="--target=riscv64-linux-gnu -march=rv64gcv_zvbb" CXXFLAGS="${CFLAGS}" \
|
||||
cmake --toolchain=cmake/toolchains-ci/riscv64-linux-gnu.cmake -DSIMDJSON_DEVELOPER_MODE=ON -B build
|
||||
cmake --build build/ -j$(nproc) --config Release
|
||||
- name: Test VLEN=1024
|
||||
run: |
|
||||
export QEMU_LD_PREFIX="/usr/riscv64-linux-gnu"
|
||||
export QEMU_CPU="rv64,v=on,zvbb=on,vlen=1024,rvv_ta_all_1s=on,rvv_ma_all_1s=on"
|
||||
QEMU_LD_PREFIX="/usr/riscv64-linux-gnu" \
|
||||
QEMU_CPU="rv64,v=on,zvbb=on,vlen=1024,rvv_ta_all_1s=on,rvv_ma_all_1s=on" \
|
||||
ctest --timeout 1800 --output-on-failure --test-dir build -j $(nproc)
|
||||
|
||||
@@ -19,11 +19,6 @@ jobs:
|
||||
sudo apt-get install -y cmake make g++-riscv64-linux-gnu qemu-user-static clang-17
|
||||
- name: Build
|
||||
run: |
|
||||
CXX=clang++-17 CXXFLAGS="--target=riscv64-linux-gnu -march=rv64gcv" \
|
||||
cmake --toolchain=cmake/toolchains-ci/riscv64-linux-gnu.cmake -DCMAKE_BUILD_TYPE=Release -B build
|
||||
cmake --build build/ -j$(nproc)
|
||||
- name: Test VLEN=128
|
||||
run: |
|
||||
export QEMU_LD_PREFIX="/usr/riscv64-linux-gnu"
|
||||
export QEMU_CPU="rv64,v=on,vlen=128,rvv_ta_all_1s=on,rvv_ma_all_1s=on"
|
||||
ctest --timeout 1800 --output-on-failure --test-dir build -j $(nproc)
|
||||
CC=clang-17 CXX=clang++-17 CFLAGS="--target=riscv64-linux-gnu -march=rv64gcv" CXXFLAGS="${CFLAGS}" \
|
||||
cmake --toolchain=cmake/toolchains-ci/riscv64-linux-gnu.cmake -DSIMDJSON_DEVELOPER_MODE=ON -B build
|
||||
cmake --build build/ -j$(nproc) --config Release
|
||||
|
||||
@@ -0,0 +1,39 @@
|
||||
name: Ubuntu rvv VLEN=128 (clang 20)
|
||||
|
||||
on:
|
||||
push:
|
||||
branches:
|
||||
- master
|
||||
pull_request:
|
||||
branches:
|
||||
- master
|
||||
|
||||
jobs:
|
||||
build:
|
||||
runs-on: ubuntu-24.04
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- name: Install packages
|
||||
run: |
|
||||
sudo apt-get update -q -y
|
||||
sudo apt-get install -y cmake make g++-riscv64-linux-gnu qemu-user-static clang-20
|
||||
- name: Build
|
||||
run: |
|
||||
CC=clang-20 CXX=clang++-20 CFLAGS="--target=riscv64-linux-gnu -march=rv64gcv" CXXFLAGS="${CFLAGS}" \
|
||||
cmake --toolchain=cmake/toolchains-ci/riscv64-linux-gnu.cmake -DSIMDJSON_DEVELOPER_MODE=ON -B build
|
||||
cmake --build build/ -j$(nproc) --config Release
|
||||
- name: Test VLEN=128
|
||||
run: |
|
||||
QEMU_LD_PREFIX="/usr/riscv64-linux-gnu" \
|
||||
QEMU_CPU="rv64,v=on,vlen=128,rvv_ta_all_1s=on,rvv_ma_all_1s=on" \
|
||||
ctest --timeout 1800 --output-on-failure --test-dir build -j $(nproc)
|
||||
- name: Build VLS
|
||||
run: |
|
||||
CC=clang-20 CXX=clang++-20 CFLAGS="--target=riscv64-linux-gnu -march=rv64gcv_zvl128b_zba_zbb_zbc -mrvv-vector-bits=zvl" CXXFLAGS="${CFLAGS}" \
|
||||
cmake --toolchain=cmake/toolchains-ci/riscv64-linux-gnu.cmake -DSIMDJSON_DEVELOPER_MODE=ON -B build-vls
|
||||
cmake --build build-vls/ -j$(nproc) --config Release
|
||||
- name: Test VLEN=128 VLS
|
||||
run: |
|
||||
QEMU_LD_PREFIX="/usr/riscv64-linux-gnu" \
|
||||
QEMU_CPU="rv64,v=on,zba=on,zbb=on,zbc=on,vlen=128,rvv_ta_all_1s=on,rvv_ma_all_1s=on" \
|
||||
ctest --timeout 1800 --output-on-failure --test-dir build-vls -j $(nproc)
|
||||
@@ -19,11 +19,21 @@ jobs:
|
||||
sudo apt-get install -y cmake make g++-14-riscv64-linux-gnu qemu-user-static
|
||||
- name: Build
|
||||
run: |
|
||||
CXX=riscv64-linux-gnu-g++-14 CXXFLAGS=-march=rv64gcv \
|
||||
cmake --toolchain=cmake/toolchains-ci/riscv64-linux-gnu.cmake -DCMAKE_BUILD_TYPE=Release -B build
|
||||
cmake --build build/ -j$(nproc)
|
||||
CC=riscv64-linux-gnu-gcc-14 CXX=riscv64-linux-gnu-g++-14 CFLAGS=-march=rv64gcv CXXFLAGS="${CFLAGS}" \
|
||||
cmake --toolchain=cmake/toolchains-ci/riscv64-linux-gnu.cmake -DSIMDJSON_DEVELOPER_MODE=ON -B build
|
||||
cmake --build build/ -j$(nproc) --config Release
|
||||
- name: Test VLEN=256
|
||||
run: |
|
||||
export QEMU_LD_PREFIX="/usr/riscv64-linux-gnu"
|
||||
export QEMU_CPU="rv64,v=on,zvbb=on,vlen=256,rvv_ta_all_1s=on,rvv_ma_all_1s=on"
|
||||
QEMU_LD_PREFIX="/usr/riscv64-linux-gnu" \
|
||||
QEMU_CPU="rv64,v=on,zvbb=on,vlen=256,rvv_ta_all_1s=on,rvv_ma_all_1s=on" \
|
||||
ctest --timeout 1800 --output-on-failure --test-dir build -j $(nproc)
|
||||
- name: Build VLS
|
||||
run: |
|
||||
CC=riscv64-linux-gnu-gcc-14 CXX=riscv64-linux-gnu-g++-14 CFLAGS="-march=rv64gcv_zvl256b -mrvv-vector-bits=zvl" CXXFLAGS="${CFLAGS}" \
|
||||
cmake --toolchain=cmake/toolchains-ci/riscv64-linux-gnu.cmake -DSIMDJSON_DEVELOPER_MODE=ON -B build-vls
|
||||
cmake --build build-vls/ -j$(nproc) --config Release
|
||||
- name: Test VLEN=256 VLS
|
||||
run: |
|
||||
QEMU_LD_PREFIX="/usr/riscv64-linux-gnu" \
|
||||
QEMU_CPU="rv64,v=on,zvbb=on,vlen=256,rvv_ta_all_1s=on,rvv_ma_all_1s=on" \
|
||||
ctest --timeout 1800 --output-on-failure --test-dir build-vls -j $(nproc)
|
||||
|
||||
@@ -0,0 +1,39 @@
|
||||
name: Ubuntu rvv VLEN=512 (clang 19)
|
||||
|
||||
on:
|
||||
push:
|
||||
branches:
|
||||
- master
|
||||
pull_request:
|
||||
branches:
|
||||
- master
|
||||
|
||||
jobs:
|
||||
build:
|
||||
runs-on: ubuntu-24.04
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- name: Install packages
|
||||
run: |
|
||||
sudo apt-get update -q -y
|
||||
sudo apt-get install -y cmake make g++-riscv64-linux-gnu qemu-user-static clang-19
|
||||
- name: Build
|
||||
run: |
|
||||
CC=clang-19 CXX=clang++-19 CFLAGS="--target=riscv64-linux-gnu -march=rv64gcv" CXXFLAGS="${CFLAGS}" \
|
||||
cmake --toolchain=cmake/toolchains-ci/riscv64-linux-gnu.cmake -DSIMDJSON_DEVELOPER_MODE=ON -B build
|
||||
cmake --build build/ -j$(nproc) --config Release
|
||||
- name: Test VLEN=512
|
||||
run: |
|
||||
QEMU_LD_PREFIX="/usr/riscv64-linux-gnu" \
|
||||
QEMU_CPU="rv64,v=on,vlen=512,rvv_ta_all_1s=on,rvv_ma_all_1s=on" \
|
||||
ctest --timeout 1800 --output-on-failure --test-dir build -j $(nproc)
|
||||
- name: Build VLS
|
||||
run: |
|
||||
CC=clang-19 CXX=clang++-19 CFLAGS="--target=riscv64-linux-gnu -march=rv64gcv_zvl512b_zba_zbb_zbc -mrvv-vector-bits=zvl" CXXFLAGS="${CFLAGS}" \
|
||||
cmake --toolchain=cmake/toolchains-ci/riscv64-linux-gnu.cmake -DSIMDJSON_DEVELOPER_MODE=ON -B build-vls
|
||||
cmake --build build-vls/ -j$(nproc) --config Release
|
||||
- name: Test VLEN=512 VLS
|
||||
run: |
|
||||
QEMU_LD_PREFIX="/usr/riscv64-linux-gnu" \
|
||||
QEMU_CPU="rv64,v=on,zba=on,zbb=on,zbc=on,vlen=512,rvv_ta_all_1s=on,rvv_ma_all_1s=on" \
|
||||
ctest --timeout 1800 --output-on-failure --test-dir build-vls -j $(nproc)
|
||||
+3
-3
@@ -12,7 +12,7 @@ endif()
|
||||
project(
|
||||
simdjson
|
||||
# The version number is modified by tools/release.py
|
||||
VERSION 4.2.4
|
||||
VERSION 4.3.0
|
||||
DESCRIPTION "Parsing gigabytes of JSON per second"
|
||||
HOMEPAGE_URL "https://simdjson.org/"
|
||||
LANGUAGES CXX C
|
||||
@@ -29,8 +29,8 @@ string(
|
||||
# ---- Options, variables ----
|
||||
|
||||
# These version numbers are modified by tools/release.py
|
||||
set(SIMDJSON_LIB_VERSION "29.0.0" CACHE STRING "simdjson library version")
|
||||
set(SIMDJSON_LIB_SOVERSION "29" CACHE STRING "simdjson library soversion")
|
||||
set(SIMDJSON_LIB_VERSION "30.0.0" CACHE STRING "simdjson library version")
|
||||
set(SIMDJSON_LIB_SOVERSION "30" CACHE STRING "simdjson library soversion")
|
||||
|
||||
option(SIMDJSON_BUILD_STATIC_LIB "Build simdjson_static library along with simdjson (only makes sense if BUILD_SHARED_LIBS=ON)" OFF)
|
||||
if(SIMDJSON_BUILD_STATIC_LIB AND NOT BUILD_SHARED_LIBS)
|
||||
|
||||
@@ -38,7 +38,7 @@ PROJECT_NAME = simdjson
|
||||
# could be handy for archiving the generated documentation or if some version
|
||||
# control system is used.
|
||||
|
||||
PROJECT_NUMBER = "4.2.4"
|
||||
PROJECT_NUMBER = "4.3.0"
|
||||
|
||||
# Using the PROJECT_BRIEF tag one can provide an optional one line description
|
||||
# for a project that appears at the top of each page and should give viewer a
|
||||
|
||||
@@ -64,6 +64,7 @@ Real-world usage
|
||||
- [RonDB](https://github.com/logicalclocks/rondb)
|
||||
- [GreptimeDB](https://github.com/GreptimeTeam/greptimedb)
|
||||
- [mamba](https://github.com/mamba-org/mamba)
|
||||
- [Ladybird Browser](https://ladybird.org)
|
||||
|
||||
|
||||
If you are planning to use simdjson in a product, please work from one of our releases.
|
||||
|
||||
@@ -8,7 +8,7 @@ CPMAddPackage(
|
||||
|
||||
option(SIMDJSON_USE_RUST "Build the static_reflect benchmark" OFF)
|
||||
|
||||
if(SIMDJSON_USER_RUST)
|
||||
if(SIMDJSON_USE_RUST)
|
||||
if(NOT WIN32)
|
||||
# We want the check whether Rust is available before trying to build a crate.
|
||||
CPMAddPackage(
|
||||
@@ -39,9 +39,9 @@ if(SIMDJSON_USER_RUST)
|
||||
message(STATUS "curl https://sh.rustup.rs -sSf | sh")
|
||||
endif()
|
||||
endif()
|
||||
else(SIMDJSON_USER_RUST)
|
||||
else(SIMDJSON_USE_RUST)
|
||||
message(STATUS "We will not benchmark serde-benchmark." )
|
||||
endif(SIMDJSON_USER_RUST)
|
||||
endif(SIMDJSON_USE_RUST)
|
||||
|
||||
# Add the benchmark executable targets
|
||||
add_subdirectory(twitter_benchmark)
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
add_executable(benchmark_serialization_citm_catalog benchmark_serialization_citm_catalog.cpp)
|
||||
add_executable(benchmark_parsing_citm benchmark_parsing_citm.cpp)
|
||||
|
||||
# Link with Rust benchmarking code if available
|
||||
if(TARGET serde-benchmark)
|
||||
@@ -11,4 +12,29 @@ target_link_libraries(benchmark_serialization_citm_catalog PRIVATE simdjson::sim
|
||||
target_link_libraries(benchmark_serialization_citm_catalog PRIVATE reflectcpp)
|
||||
target_compile_definitions(benchmark_serialization_citm_catalog PRIVATE SIMDJSON_BENCH_CPP_REFLECT=1)
|
||||
|
||||
target_compile_definitions(benchmark_serialization_citm_catalog PRIVATE JSON_FILE="${BENCH_CITM_JSON}")
|
||||
if(TARGET yyjson)
|
||||
target_link_libraries(benchmark_serialization_citm_catalog PRIVATE yyjson)
|
||||
target_compile_definitions(benchmark_serialization_citm_catalog PRIVATE SIMDJSON_COMPETITION_YYJSON)
|
||||
endif()
|
||||
|
||||
target_compile_definitions(benchmark_serialization_citm_catalog PRIVATE JSON_FILE="${BENCH_CITM_JSON}")
|
||||
|
||||
# Configuration for parsing benchmark
|
||||
if(TARGET serde-benchmark)
|
||||
target_link_libraries(benchmark_parsing_citm PRIVATE serde-benchmark)
|
||||
target_compile_definitions(benchmark_parsing_citm PRIVATE SIMDJSON_RUST_VERSION="${Rust_VERSION}")
|
||||
endif()
|
||||
|
||||
target_link_libraries(benchmark_parsing_citm PRIVATE simdjson::simdjson nlohmann_json)
|
||||
|
||||
if(TARGET rapidjson)
|
||||
target_link_libraries(benchmark_parsing_citm PRIVATE rapidjson)
|
||||
target_compile_definitions(benchmark_parsing_citm PRIVATE SIMDJSON_COMPETITION_RAPIDJSON)
|
||||
endif()
|
||||
|
||||
if(TARGET yyjson)
|
||||
target_link_libraries(benchmark_parsing_citm PRIVATE yyjson)
|
||||
target_compile_definitions(benchmark_parsing_citm PRIVATE SIMDJSON_COMPETITION_YYJSON)
|
||||
endif()
|
||||
|
||||
target_compile_definitions(benchmark_parsing_citm PRIVATE JSON_FILE="${BENCH_CITM_JSON}")
|
||||
@@ -0,0 +1,563 @@
|
||||
#include <cassert>
|
||||
#include <cstdlib>
|
||||
#include <ctime>
|
||||
#include <format>
|
||||
#include <fstream>
|
||||
#include <iostream>
|
||||
#include <nlohmann/json.hpp>
|
||||
#include <simdjson.h>
|
||||
#include <string>
|
||||
#include "citm_catalog_data.h"
|
||||
#include "nlohmann_citm_catalog_data.h"
|
||||
#include "../benchmark_utils/benchmark_helper.h"
|
||||
|
||||
#ifdef SIMDJSON_COMPETITION_RAPIDJSON
|
||||
#include "rapidjson_citm_catalog_data.h"
|
||||
#endif
|
||||
|
||||
#ifdef SIMDJSON_COMPETITION_YYJSON
|
||||
#include "yyjson_citm_catalog_data.h"
|
||||
#endif
|
||||
|
||||
#ifdef SIMDJSON_RUST_VERSION
|
||||
#include "../serde-benchmark/serde_benchmark.h"
|
||||
|
||||
void bench_rust_parsing(const std::string &json_str) {
|
||||
size_t input_volume = json_str.size();
|
||||
printf("# input volume: %zu bytes\n", input_volume);
|
||||
|
||||
volatile bool result = true;
|
||||
pretty_print(1, input_volume, "bench_rust_parsing",
|
||||
bench([&json_str, &result]() {
|
||||
serde_benchmark::CitmCatalog *catalog = serde_benchmark::citm_from_str(json_str.c_str(), json_str.size());
|
||||
result = (catalog != nullptr);
|
||||
if (catalog) {
|
||||
serde_benchmark::free_citm(catalog);
|
||||
}
|
||||
if (!result) {
|
||||
printf("parse error\n");
|
||||
}
|
||||
}));
|
||||
}
|
||||
#endif
|
||||
|
||||
template <class T> void bench_simdjson_static_reflection_parsing(const std::string &json_str) {
|
||||
size_t input_volume = json_str.size();
|
||||
printf("# input volume: %zu bytes\n", input_volume);
|
||||
|
||||
// Pre-allocate padded buffer outside the benchmark loop
|
||||
std::string mutable_json = json_str;
|
||||
simdjson::pad(mutable_json);
|
||||
|
||||
volatile bool result = true;
|
||||
pretty_print(1, input_volume, "bench_simdjson_static_reflection_parsing",
|
||||
bench([&mutable_json, &result]() {
|
||||
simdjson::ondemand::parser parser;
|
||||
simdjson::ondemand::document doc;
|
||||
if(parser.iterate(mutable_json).get(doc)) {
|
||||
result = false;
|
||||
return;
|
||||
}
|
||||
T my_struct;
|
||||
if(doc.get<T>().get(my_struct)) {
|
||||
result = false;
|
||||
}
|
||||
if (!result) {
|
||||
printf("parse error\n");
|
||||
}
|
||||
}));
|
||||
}
|
||||
|
||||
#if SIMDJSON_STATIC_REFLECTION
|
||||
template <class T> void bench_simdjson_from_parsing(const std::string &json_str) {
|
||||
size_t input_volume = json_str.size();
|
||||
printf("# input volume: %zu bytes\n", input_volume);
|
||||
|
||||
// Pre-allocate padded buffer outside the benchmark loop
|
||||
simdjson::padded_string padded = simdjson::padded_string(json_str);
|
||||
|
||||
volatile bool result = true;
|
||||
pretty_print(1, input_volume, "bench_simdjson_from_parsing",
|
||||
bench([&padded, &result]() {
|
||||
T my_struct;
|
||||
auto err = simdjson::from(padded).get(my_struct);
|
||||
if (err) {
|
||||
result = false;
|
||||
printf("parse error: %s\n", simdjson::error_message(err));
|
||||
return;
|
||||
}
|
||||
}));
|
||||
}
|
||||
#endif
|
||||
|
||||
// nlohmann::json deserialization functions
|
||||
void from_json(const nlohmann::json &j, CITMPrice &p) {
|
||||
j.at("amount").get_to(p.amount);
|
||||
j.at("audienceSubCategoryId").get_to(p.audienceSubCategoryId);
|
||||
j.at("seatCategoryId").get_to(p.seatCategoryId);
|
||||
}
|
||||
|
||||
void from_json(const nlohmann::json &j, CITMArea &a) {
|
||||
j.at("areaId").get_to(a.areaId);
|
||||
j.at("blockIds").get_to(a.blockIds);
|
||||
}
|
||||
|
||||
void from_json(const nlohmann::json &j, CITMSeatCategory &s) {
|
||||
j.at("areas").get_to(s.areas);
|
||||
j.at("seatCategoryId").get_to(s.seatCategoryId);
|
||||
}
|
||||
|
||||
void from_json(const nlohmann::json &j, CITMPerformance &p) {
|
||||
j.at("id").get_to(p.id);
|
||||
j.at("eventId").get_to(p.eventId);
|
||||
if (j.contains("logo") && !j["logo"].is_null()) {
|
||||
p.logo = j["logo"].get<std::string>();
|
||||
}
|
||||
if (j.contains("name") && !j["name"].is_null()) {
|
||||
p.name = j["name"].get<std::string>();
|
||||
}
|
||||
j.at("prices").get_to(p.prices);
|
||||
j.at("seatCategories").get_to(p.seatCategories);
|
||||
if (j.contains("seatMapImage") && !j["seatMapImage"].is_null()) {
|
||||
p.seatMapImage = j["seatMapImage"].get<std::string>();
|
||||
}
|
||||
j.at("start").get_to(p.start);
|
||||
j.at("venueCode").get_to(p.venueCode);
|
||||
}
|
||||
|
||||
void from_json(const nlohmann::json &j, CITMEvent &e) {
|
||||
j.at("id").get_to(e.id);
|
||||
j.at("name").get_to(e.name);
|
||||
if (j.contains("description") && !j["description"].is_null()) {
|
||||
e.description = j["description"].get<std::string>();
|
||||
}
|
||||
if (j.contains("logo") && !j["logo"].is_null()) {
|
||||
e.logo = j["logo"].get<std::string>();
|
||||
}
|
||||
j.at("subTopicIds").get_to(e.subTopicIds);
|
||||
if (j.contains("subjectCode") && !j["subjectCode"].is_null()) {
|
||||
e.subjectCode = j["subjectCode"].get<std::string>();
|
||||
}
|
||||
if (j.contains("subtitle") && !j["subtitle"].is_null()) {
|
||||
e.subtitle = j["subtitle"].get<std::string>();
|
||||
}
|
||||
j.at("topicIds").get_to(e.topicIds);
|
||||
}
|
||||
|
||||
void from_json(const nlohmann::json &j, CitmCatalog &c) {
|
||||
j.at("events").get_to(c.events);
|
||||
j.at("performances").get_to(c.performances);
|
||||
}
|
||||
|
||||
CitmCatalog nlohmann_deserialize(const std::string &json_str) {
|
||||
nlohmann::json j = nlohmann::json::parse(json_str);
|
||||
return j.get<CitmCatalog>();
|
||||
}
|
||||
|
||||
void bench_nlohmann_parsing(const std::string &json_str) {
|
||||
size_t input_volume = json_str.size();
|
||||
printf("# input volume: %zu bytes\n", input_volume);
|
||||
|
||||
volatile bool result = true;
|
||||
pretty_print(1, input_volume, "bench_nlohmann_parsing",
|
||||
bench([&json_str, &result]() {
|
||||
try {
|
||||
CitmCatalog data = nlohmann_deserialize(json_str);
|
||||
result = true;
|
||||
} catch (...) {
|
||||
result = false;
|
||||
printf("parse error\n");
|
||||
}
|
||||
}));
|
||||
}
|
||||
|
||||
#ifdef SIMDJSON_COMPETITION_RAPIDJSON
|
||||
CitmCatalog rapidjson_deserialize(const std::string &json_str) {
|
||||
rapidjson::Document doc;
|
||||
doc.Parse(json_str.c_str());
|
||||
|
||||
if (doc.HasParseError()) {
|
||||
throw std::runtime_error("RapidJSON parse error");
|
||||
}
|
||||
|
||||
CitmCatalog catalog;
|
||||
|
||||
// Parse events
|
||||
if (doc.HasMember("events") && doc["events"].IsObject()) {
|
||||
for (auto& m : doc["events"].GetObject()) {
|
||||
CITMEvent event;
|
||||
const auto& e = m.value;
|
||||
|
||||
event.id = e["id"].GetUint64();
|
||||
event.name = e["name"].GetString();
|
||||
if (e.HasMember("description") && !e["description"].IsNull()) {
|
||||
event.description = e["description"].GetString();
|
||||
}
|
||||
if (e.HasMember("logo") && !e["logo"].IsNull()) {
|
||||
event.logo = e["logo"].GetString();
|
||||
}
|
||||
|
||||
event.subTopicIds.clear();
|
||||
for (auto& id : e["subTopicIds"].GetArray()) {
|
||||
event.subTopicIds.push_back(id.GetUint64());
|
||||
}
|
||||
|
||||
if (e.HasMember("subjectCode") && !e["subjectCode"].IsNull()) {
|
||||
event.subjectCode = e["subjectCode"].GetString();
|
||||
}
|
||||
if (e.HasMember("subtitle") && !e["subtitle"].IsNull()) {
|
||||
event.subtitle = e["subtitle"].GetString();
|
||||
}
|
||||
|
||||
event.topicIds.clear();
|
||||
for (auto& id : e["topicIds"].GetArray()) {
|
||||
event.topicIds.push_back(id.GetUint64());
|
||||
}
|
||||
|
||||
catalog.events[m.name.GetString()] = event;
|
||||
}
|
||||
}
|
||||
|
||||
// Parse performances
|
||||
if (doc.HasMember("performances") && doc["performances"].IsArray()) {
|
||||
for (auto& p : doc["performances"].GetArray()) {
|
||||
CITMPerformance perf;
|
||||
|
||||
perf.id = p["id"].GetUint64();
|
||||
perf.eventId = p["eventId"].GetUint64();
|
||||
if (p.HasMember("logo") && !p["logo"].IsNull()) {
|
||||
perf.logo = p["logo"].GetString();
|
||||
}
|
||||
if (p.HasMember("name") && !p["name"].IsNull()) {
|
||||
perf.name = p["name"].GetString();
|
||||
}
|
||||
|
||||
// Parse prices
|
||||
for (auto& price : p["prices"].GetArray()) {
|
||||
CITMPrice pr;
|
||||
pr.amount = price["amount"].GetUint64();
|
||||
pr.audienceSubCategoryId = price["audienceSubCategoryId"].GetUint64();
|
||||
pr.seatCategoryId = price["seatCategoryId"].GetUint64();
|
||||
perf.prices.push_back(pr);
|
||||
}
|
||||
|
||||
// Parse seat categories
|
||||
for (auto& sc : p["seatCategories"].GetArray()) {
|
||||
CITMSeatCategory seatCat;
|
||||
seatCat.seatCategoryId = sc["seatCategoryId"].GetUint64();
|
||||
|
||||
for (auto& area : sc["areas"].GetArray()) {
|
||||
CITMArea ar;
|
||||
ar.areaId = area["areaId"].GetUint64();
|
||||
for (auto& block : area["blockIds"].GetArray()) {
|
||||
ar.blockIds.push_back(block.GetUint64());
|
||||
}
|
||||
seatCat.areas.push_back(ar);
|
||||
}
|
||||
perf.seatCategories.push_back(seatCat);
|
||||
}
|
||||
|
||||
if (p.HasMember("seatMapImage") && !p["seatMapImage"].IsNull()) {
|
||||
perf.seatMapImage = p["seatMapImage"].GetString();
|
||||
}
|
||||
perf.start = p["start"].GetUint64();
|
||||
perf.venueCode = p["venueCode"].GetString();
|
||||
|
||||
catalog.performances.push_back(perf);
|
||||
}
|
||||
}
|
||||
|
||||
return catalog;
|
||||
}
|
||||
|
||||
void bench_rapidjson_parsing(const std::string &json_str) {
|
||||
size_t input_volume = json_str.size();
|
||||
printf("# input volume: %zu bytes\n", input_volume);
|
||||
|
||||
volatile bool result = true;
|
||||
pretty_print(1, input_volume, "bench_rapidjson_parsing",
|
||||
bench([&json_str, &result]() {
|
||||
try {
|
||||
CitmCatalog data = rapidjson_deserialize(json_str);
|
||||
result = true;
|
||||
} catch (...) {
|
||||
result = false;
|
||||
printf("parse error\n");
|
||||
}
|
||||
}));
|
||||
}
|
||||
#endif
|
||||
|
||||
#ifdef SIMDJSON_COMPETITION_YYJSON
|
||||
CitmCatalog yyjson_deserialize(const std::string &json_str) {
|
||||
yyjson_doc *doc = yyjson_read(json_str.c_str(), json_str.size(), 0);
|
||||
if (!doc) {
|
||||
throw std::runtime_error("YYJson parse error");
|
||||
}
|
||||
|
||||
yyjson_val *root = yyjson_doc_get_root(doc);
|
||||
CitmCatalog catalog;
|
||||
|
||||
// Parse events
|
||||
yyjson_val *events = yyjson_obj_get(root, "events");
|
||||
if (events) {
|
||||
size_t idx, max;
|
||||
yyjson_val *key, *val;
|
||||
yyjson_obj_foreach(events, idx, max, key, val) {
|
||||
CITMEvent event;
|
||||
|
||||
event.id = yyjson_get_uint(yyjson_obj_get(val, "id"));
|
||||
const char* name = yyjson_get_str(yyjson_obj_get(val, "name"));
|
||||
if (name) event.name = name;
|
||||
|
||||
yyjson_val *desc = yyjson_obj_get(val, "description");
|
||||
if (desc && !yyjson_is_null(desc)) {
|
||||
const char* str = yyjson_get_str(desc);
|
||||
if (str) event.description = str;
|
||||
}
|
||||
|
||||
yyjson_val *logo = yyjson_obj_get(val, "logo");
|
||||
if (logo && !yyjson_is_null(logo)) {
|
||||
const char* str = yyjson_get_str(logo);
|
||||
if (str) event.logo = str;
|
||||
}
|
||||
|
||||
yyjson_val *subTopics = yyjson_obj_get(val, "subTopicIds");
|
||||
if (subTopics) {
|
||||
size_t sidx, smax;
|
||||
yyjson_val *sval;
|
||||
yyjson_arr_foreach(subTopics, sidx, smax, sval) {
|
||||
event.subTopicIds.push_back(yyjson_get_uint(sval));
|
||||
}
|
||||
}
|
||||
|
||||
yyjson_val *subjectCode = yyjson_obj_get(val, "subjectCode");
|
||||
if (subjectCode && !yyjson_is_null(subjectCode)) {
|
||||
const char* str = yyjson_get_str(subjectCode);
|
||||
if (str) event.subjectCode = str;
|
||||
}
|
||||
|
||||
yyjson_val *subtitle = yyjson_obj_get(val, "subtitle");
|
||||
if (subtitle && !yyjson_is_null(subtitle)) {
|
||||
const char* str = yyjson_get_str(subtitle);
|
||||
if (str) event.subtitle = str;
|
||||
}
|
||||
|
||||
yyjson_val *topics = yyjson_obj_get(val, "topicIds");
|
||||
if (topics) {
|
||||
size_t tidx, tmax;
|
||||
yyjson_val *tval;
|
||||
yyjson_arr_foreach(topics, tidx, tmax, tval) {
|
||||
event.topicIds.push_back(yyjson_get_uint(tval));
|
||||
}
|
||||
}
|
||||
|
||||
const char* keyStr = yyjson_get_str(key);
|
||||
if (keyStr) {
|
||||
catalog.events[keyStr] = event;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Parse performances
|
||||
yyjson_val *performances = yyjson_obj_get(root, "performances");
|
||||
if (performances) {
|
||||
size_t idx, max;
|
||||
yyjson_val *val;
|
||||
yyjson_arr_foreach(performances, idx, max, val) {
|
||||
CITMPerformance perf;
|
||||
|
||||
perf.id = yyjson_get_uint(yyjson_obj_get(val, "id"));
|
||||
perf.eventId = yyjson_get_uint(yyjson_obj_get(val, "eventId"));
|
||||
|
||||
yyjson_val *logo = yyjson_obj_get(val, "logo");
|
||||
if (logo && !yyjson_is_null(logo)) {
|
||||
const char* str = yyjson_get_str(logo);
|
||||
if (str) perf.logo = str;
|
||||
}
|
||||
|
||||
yyjson_val *name = yyjson_obj_get(val, "name");
|
||||
if (name && !yyjson_is_null(name)) {
|
||||
const char* str = yyjson_get_str(name);
|
||||
if (str) perf.name = str;
|
||||
}
|
||||
|
||||
// Parse prices
|
||||
yyjson_val *prices = yyjson_obj_get(val, "prices");
|
||||
if (prices) {
|
||||
size_t pidx, pmax;
|
||||
yyjson_val *pval;
|
||||
yyjson_arr_foreach(prices, pidx, pmax, pval) {
|
||||
CITMPrice price;
|
||||
price.amount = yyjson_get_uint(yyjson_obj_get(pval, "amount"));
|
||||
price.audienceSubCategoryId = yyjson_get_uint(yyjson_obj_get(pval, "audienceSubCategoryId"));
|
||||
price.seatCategoryId = yyjson_get_uint(yyjson_obj_get(pval, "seatCategoryId"));
|
||||
perf.prices.push_back(price);
|
||||
}
|
||||
}
|
||||
|
||||
// Parse seat categories
|
||||
yyjson_val *seatCats = yyjson_obj_get(val, "seatCategories");
|
||||
if (seatCats) {
|
||||
size_t scidx, scmax;
|
||||
yyjson_val *scval;
|
||||
yyjson_arr_foreach(seatCats, scidx, scmax, scval) {
|
||||
CITMSeatCategory seatCat;
|
||||
seatCat.seatCategoryId = yyjson_get_uint(yyjson_obj_get(scval, "seatCategoryId"));
|
||||
|
||||
yyjson_val *areas = yyjson_obj_get(scval, "areas");
|
||||
if (areas) {
|
||||
size_t aidx, amax;
|
||||
yyjson_val *aval;
|
||||
yyjson_arr_foreach(areas, aidx, amax, aval) {
|
||||
CITMArea area;
|
||||
area.areaId = yyjson_get_uint(yyjson_obj_get(aval, "areaId"));
|
||||
|
||||
yyjson_val *blocks = yyjson_obj_get(aval, "blockIds");
|
||||
if (blocks) {
|
||||
size_t bidx, bmax;
|
||||
yyjson_val *bval;
|
||||
yyjson_arr_foreach(blocks, bidx, bmax, bval) {
|
||||
area.blockIds.push_back(yyjson_get_uint(bval));
|
||||
}
|
||||
}
|
||||
seatCat.areas.push_back(area);
|
||||
}
|
||||
}
|
||||
perf.seatCategories.push_back(seatCat);
|
||||
}
|
||||
}
|
||||
|
||||
yyjson_val *seatMapImage = yyjson_obj_get(val, "seatMapImage");
|
||||
if (seatMapImage && !yyjson_is_null(seatMapImage)) {
|
||||
const char* str = yyjson_get_str(seatMapImage);
|
||||
if (str) perf.seatMapImage = str;
|
||||
}
|
||||
|
||||
perf.start = yyjson_get_uint(yyjson_obj_get(val, "start"));
|
||||
const char* venueCode = yyjson_get_str(yyjson_obj_get(val, "venueCode"));
|
||||
if (venueCode) perf.venueCode = venueCode;
|
||||
|
||||
catalog.performances.push_back(perf);
|
||||
}
|
||||
}
|
||||
|
||||
yyjson_doc_free(doc);
|
||||
return catalog;
|
||||
}
|
||||
|
||||
void bench_yyjson_parsing(const std::string &json_str) {
|
||||
size_t input_volume = json_str.size();
|
||||
printf("# input volume: %zu bytes\n", input_volume);
|
||||
|
||||
volatile bool result = true;
|
||||
pretty_print(1, input_volume, "bench_yyjson_parsing",
|
||||
bench([&json_str, &result]() {
|
||||
try {
|
||||
CitmCatalog data = yyjson_deserialize(json_str);
|
||||
result = true;
|
||||
} catch (...) {
|
||||
result = false;
|
||||
printf("parse error\n");
|
||||
}
|
||||
}));
|
||||
}
|
||||
#endif
|
||||
|
||||
std::string read_file(std::string filename) {
|
||||
printf("# Reading file %s\n", filename.c_str());
|
||||
constexpr size_t read_size = 4096;
|
||||
auto stream = std::ifstream(filename);
|
||||
stream.exceptions(std::ios_base::badbit);
|
||||
|
||||
if (!stream) {
|
||||
std::cerr << "Error: Failed to open file " << filename << std::endl;
|
||||
exit(EXIT_FAILURE);
|
||||
}
|
||||
|
||||
std::string out;
|
||||
auto buf = std::string(read_size, '\0');
|
||||
while (stream.read(&buf[0], read_size)) {
|
||||
out.append(buf, 0, size_t(stream.gcount()));
|
||||
}
|
||||
out.append(buf, 0, size_t(stream.gcount()));
|
||||
return out;
|
||||
}
|
||||
|
||||
// Function to check if benchmark name matches any of the comma-separated filters
|
||||
bool matches_filter(const std::string& benchmark_name, const std::string& filter) {
|
||||
if (filter.empty()) return true;
|
||||
|
||||
// Split filter by comma
|
||||
size_t start = 0;
|
||||
size_t end = filter.find(',');
|
||||
while (end != std::string::npos) {
|
||||
std::string token = filter.substr(start, end - start);
|
||||
if (benchmark_name.find(token) != std::string::npos) {
|
||||
return true;
|
||||
}
|
||||
start = end + 1;
|
||||
end = filter.find(',', start);
|
||||
}
|
||||
// Check last token
|
||||
std::string token = filter.substr(start);
|
||||
return benchmark_name.find(token) != std::string::npos;
|
||||
}
|
||||
|
||||
int main(int argc, char *argv[]) {
|
||||
// Get the JSON file path from preprocessor or use default
|
||||
std::string filename;
|
||||
#ifdef JSON_FILE
|
||||
filename = JSON_FILE;
|
||||
#else
|
||||
filename = "jsonexamples/citm_catalog.json";
|
||||
#endif
|
||||
|
||||
std::string json_str = read_file(filename);
|
||||
|
||||
// Parse command-line arguments for filter
|
||||
std::string filter;
|
||||
for (int i = 1; i < argc; i++) {
|
||||
std::string arg = argv[i];
|
||||
if (arg == "-f" && i + 1 < argc) {
|
||||
filter = argv[i + 1];
|
||||
printf("# Filter: %s\n", filter.c_str());
|
||||
i++;
|
||||
}
|
||||
}
|
||||
|
||||
// If no filter provided, run all benchmarks
|
||||
if (filter.empty()) {
|
||||
printf("# Running all benchmarks (use -f <filter> to run specific ones)\n");
|
||||
}
|
||||
|
||||
// Benchmarking the parsing
|
||||
if (matches_filter("nlohmann", filter)) {
|
||||
bench_nlohmann_parsing(json_str);
|
||||
}
|
||||
#ifdef SIMDJSON_COMPETITION_RAPIDJSON
|
||||
if (matches_filter("rapidjson", filter)) {
|
||||
bench_rapidjson_parsing(json_str);
|
||||
}
|
||||
#endif
|
||||
#ifdef SIMDJSON_COMPETITION_YYJSON
|
||||
if (matches_filter("yyjson", filter)) {
|
||||
bench_yyjson_parsing(json_str);
|
||||
}
|
||||
#endif
|
||||
if (matches_filter("simdjson_static_reflection", filter)) {
|
||||
bench_simdjson_static_reflection_parsing<CitmCatalog>(json_str);
|
||||
}
|
||||
#if SIMDJSON_STATIC_REFLECTION
|
||||
if (matches_filter("simdjson_from", filter)) {
|
||||
bench_simdjson_from_parsing<CitmCatalog>(json_str);
|
||||
}
|
||||
#endif
|
||||
#ifdef SIMDJSON_RUST_VERSION
|
||||
if (matches_filter("rust", filter)) {
|
||||
printf("# Note: Rust/Serde parsing test\n");
|
||||
bench_rust_parsing(json_str);
|
||||
}
|
||||
#endif
|
||||
return EXIT_SUCCESS;
|
||||
}
|
||||
+160
-22
@@ -11,6 +11,10 @@
|
||||
#include "nlohmann_citm_catalog_data.h"
|
||||
#include "../benchmark_utils/benchmark_helper.h"
|
||||
|
||||
#ifdef SIMDJSON_COMPETITION_YYJSON
|
||||
#include "yyjson_citm_catalog_data.h"
|
||||
#endif
|
||||
|
||||
#if SIMDJSON_BENCH_CPP_REFLECT
|
||||
#include <rfl.hpp>
|
||||
#include <rfl/json.hpp>
|
||||
@@ -35,20 +39,15 @@ void bench_reflect_cpp(CitmCatalog &data) {
|
||||
#include "../serde-benchmark/serde_benchmark.h"
|
||||
|
||||
void bench_rust(serde_benchmark::CitmCatalog *data) {
|
||||
const char * output = serde_benchmark::str_from_citm(data);
|
||||
size_t output_volume = strlen(output);
|
||||
serde_benchmark::set_citm_data(data);
|
||||
size_t output_volume = serde_benchmark::serialize_citm_to_string();
|
||||
printf("# output volume: %zu bytes\n", output_volume);
|
||||
|
||||
volatile size_t measured_volume = 0;
|
||||
pretty_print(1, output_volume, "bench_rust",
|
||||
bench([&data, &measured_volume, &output_volume]() {
|
||||
const char * output = serde_benchmark::str_from_citm(data);
|
||||
measured_volume = strlen(output);
|
||||
if (measured_volume != output_volume) {
|
||||
printf("mismatch\n");
|
||||
}
|
||||
serde_benchmark::free_str(const_cast<char*>(output));
|
||||
bench([&measured_volume, &output_volume]() {
|
||||
measured_volume = serde_benchmark::serialize_citm_to_string();
|
||||
}));
|
||||
serde_benchmark::free_str(const_cast<char*>(output));
|
||||
}
|
||||
#endif // SIMDJSON_RUST_VERSION
|
||||
|
||||
@@ -68,7 +67,55 @@ void bench_nlohmann(CitmCatalog &data) {
|
||||
}));
|
||||
}
|
||||
|
||||
#ifdef SIMDJSON_COMPETITION_YYJSON
|
||||
void bench_yyjson(CitmCatalog &data) {
|
||||
std::string output = yyjson_serialize_citm(data);
|
||||
size_t output_volume = output.size();
|
||||
printf("# output volume: %zu bytes\n", output_volume);
|
||||
|
||||
volatile size_t measured_volume = 0;
|
||||
pretty_print(1, output_volume, "bench_yyjson",
|
||||
bench([&data, &measured_volume, &output_volume]() {
|
||||
std::string output = yyjson_serialize_citm(data);
|
||||
measured_volume = output.size();
|
||||
if (measured_volume != output_volume) {
|
||||
printf("mismatch\n");
|
||||
}
|
||||
}));
|
||||
}
|
||||
#endif
|
||||
|
||||
// Fair allocation variant: allocates fresh buffer each iteration (matches other libraries)
|
||||
void bench_simdjson_static_reflection(CitmCatalog &data) {
|
||||
// First run to determine expected size
|
||||
simdjson::builder::string_builder sb_init;
|
||||
simdjson::builder::append(sb_init, data);
|
||||
std::string_view p_init;
|
||||
if(sb_init.view().get(p_init)) {
|
||||
std::cerr << "Error!" << std::endl;
|
||||
}
|
||||
size_t output_volume = p_init.size();
|
||||
printf("# output volume: %zu bytes\n", output_volume);
|
||||
|
||||
volatile size_t measured_volume = 0;
|
||||
pretty_print(sizeof(data), output_volume, "bench_simdjson_static_reflection",
|
||||
bench([&data, &measured_volume, &output_volume]() {
|
||||
// Fresh allocation each iteration - fair comparison
|
||||
simdjson::builder::string_builder sb;
|
||||
simdjson::builder::append(sb, data);
|
||||
std::string_view p;
|
||||
if(sb.view().get(p)) {
|
||||
std::cerr << "Error!" << std::endl;
|
||||
}
|
||||
measured_volume = sb.size();
|
||||
if (measured_volume != output_volume) {
|
||||
printf("mismatch\n");
|
||||
}
|
||||
}));
|
||||
}
|
||||
|
||||
// Optimized variant: reuses buffer across iterations (shows API potential)
|
||||
void bench_simdjson_static_reflection_reuse(CitmCatalog &data) {
|
||||
simdjson::builder::string_builder sb;
|
||||
simdjson::builder::append(sb, data);
|
||||
std::string_view p;
|
||||
@@ -80,7 +127,7 @@ void bench_simdjson_static_reflection(CitmCatalog &data) {
|
||||
printf("# output volume: %zu bytes\n", output_volume);
|
||||
|
||||
volatile size_t measured_volume = 0;
|
||||
pretty_print(sizeof(data), output_volume, "bench_simdjson_static_reflection",
|
||||
pretty_print(sizeof(data), output_volume, "bench_simdjson_reuse_buffer",
|
||||
bench([&data, &measured_volume, &output_volume, &sb]() {
|
||||
sb.clear();
|
||||
simdjson::builder::append(sb, data);
|
||||
@@ -95,25 +142,97 @@ void bench_simdjson_static_reflection(CitmCatalog &data) {
|
||||
}));
|
||||
}
|
||||
|
||||
std::string read_file(const std::string &file_path, size_t read_size = 65536) {
|
||||
#if SIMDJSON_STATIC_REFLECTION
|
||||
// Fair allocation variant: allocates fresh string each iteration
|
||||
void bench_simdjson_to(CitmCatalog &data) {
|
||||
// First run to determine size
|
||||
std::string output_init;
|
||||
if (simdjson::error_code err = simdjson::builder::to_json(data, output_init); err) {
|
||||
std::cerr << "Error in to_json initialization!" << simdjson::error_message(err) << std::endl;
|
||||
return;
|
||||
}
|
||||
size_t output_volume = output_init.size();
|
||||
printf("# output volume: %zu bytes\n", output_volume);
|
||||
|
||||
volatile size_t measured_volume = 0;
|
||||
pretty_print(sizeof(data), output_volume, "bench_simdjson_to",
|
||||
bench([&data, &measured_volume, &output_volume]() {
|
||||
// Fresh allocation each iteration - fair comparison
|
||||
std::string output;
|
||||
if (simdjson::error_code err = simdjson::builder::to_json(data, output); err) {
|
||||
std::cerr << "Error in to_json!" << simdjson::error_message(err) << std::endl;
|
||||
return;
|
||||
}
|
||||
measured_volume = output.size();
|
||||
if (measured_volume != output_volume) {
|
||||
printf("mismatch\n");
|
||||
}
|
||||
}));
|
||||
}
|
||||
|
||||
// Optimized variant: reuses pre-allocated string
|
||||
void bench_simdjson_to_reuse(CitmCatalog &data) {
|
||||
std::string output;
|
||||
if (simdjson::error_code err = simdjson::builder::to_json(data, output); err) {
|
||||
std::cerr << "Error in to_json initialization!" << simdjson::error_message(err) << std::endl;
|
||||
return;
|
||||
}
|
||||
size_t output_volume = output.size();
|
||||
printf("# output volume: %zu bytes\n", output_volume);
|
||||
|
||||
// Pre-allocate string with sufficient capacity to avoid reallocation
|
||||
output.reserve(output_volume * 2);
|
||||
|
||||
volatile size_t measured_volume = 0;
|
||||
pretty_print(sizeof(data), output_volume, "bench_simdjson_to_reuse",
|
||||
bench([&data, &measured_volume, &output_volume, &output]() {
|
||||
// Reuse the pre-allocated string - avoids allocation
|
||||
if (simdjson::error_code err = simdjson::builder::to_json(data, output); err) {
|
||||
std::cerr << "Error in to_json!" << simdjson::error_message(err) << std::endl;
|
||||
return;
|
||||
}
|
||||
measured_volume = output.size();
|
||||
if (measured_volume != output_volume) {
|
||||
printf("mismatch\n");
|
||||
}
|
||||
}));
|
||||
}
|
||||
#endif
|
||||
|
||||
simdjson::padded_string read_file(const std::string &file_path, size_t read_size = 65536) {
|
||||
std::ifstream stream(file_path, std::ios::binary);
|
||||
if(!stream) {
|
||||
std::cerr << "Could not open file '" << file_path << "'" << std::endl;
|
||||
exit(EXIT_FAILURE);
|
||||
}
|
||||
stream.exceptions(std::ios_base::badbit);
|
||||
std::string out;
|
||||
simdjson::padded_string_builder builder;
|
||||
std::string buf(read_size, '\0');
|
||||
while (stream.read(&buf[0], read_size)) {
|
||||
out.append(buf, 0, size_t(stream.gcount()));
|
||||
builder.append(buf.data(), size_t(stream.gcount()));
|
||||
}
|
||||
out.append(buf, 0, size_t(stream.gcount()));
|
||||
return out;
|
||||
builder.append(buf.data(), size_t(stream.gcount()));
|
||||
return builder.convert();
|
||||
}
|
||||
|
||||
// Function to check if benchmark name contains filter substring
|
||||
// Function to check if benchmark name matches any of the comma-separated filters
|
||||
bool matches_filter(const std::string& benchmark_name, const std::string& filter) {
|
||||
return filter.empty() || benchmark_name.find(filter) != std::string::npos;
|
||||
if (filter.empty()) return true;
|
||||
|
||||
// Split filter by comma
|
||||
size_t start = 0;
|
||||
size_t end = filter.find(',');
|
||||
while (end != std::string::npos) {
|
||||
std::string token = filter.substr(start, end - start);
|
||||
if (benchmark_name.find(token) != std::string::npos) {
|
||||
return true;
|
||||
}
|
||||
start = end + 1;
|
||||
end = filter.find(',', start);
|
||||
}
|
||||
// Check last token
|
||||
std::string token = filter.substr(start);
|
||||
return benchmark_name.find(token) != std::string::npos;
|
||||
}
|
||||
|
||||
int main(int argc, char* argv[]) {
|
||||
@@ -131,12 +250,12 @@ int main(int argc, char* argv[]) {
|
||||
}
|
||||
}
|
||||
// Testing correctness of round-trip (serialization + deserialization)
|
||||
std::string json_str = read_file(JSON_FILE);
|
||||
simdjson::padded_string json_str = read_file(JSON_FILE);
|
||||
|
||||
// Loading up the data into a structure.
|
||||
simdjson::ondemand::parser parser;
|
||||
simdjson::ondemand::document doc;
|
||||
if(parser.iterate(simdjson::pad(json_str)).get(doc)) {
|
||||
if(parser.iterate(json_str).get(doc)) {
|
||||
std::cerr << "Error loading the document!" << std::endl;
|
||||
return EXIT_FAILURE;
|
||||
}
|
||||
@@ -147,18 +266,37 @@ int main(int argc, char* argv[]) {
|
||||
}
|
||||
|
||||
// Benchmarking the serialization
|
||||
// Note: simdjson benchmarks include both "fair" (fresh allocation) and "reuse" (buffer reuse) variants
|
||||
// The "fair" variants allocate fresh memory each iteration, matching other libraries' behavior
|
||||
// The "reuse" variants demonstrate the API's potential when buffer reuse is possible
|
||||
|
||||
if (matches_filter("nlohmann", filter)) {
|
||||
bench_nlohmann(my_struct);
|
||||
}
|
||||
#ifdef SIMDJSON_COMPETITION_YYJSON
|
||||
if (matches_filter("yyjson", filter)) {
|
||||
bench_yyjson(my_struct);
|
||||
}
|
||||
#endif
|
||||
if (matches_filter("simdjson_static_reflection", filter)) {
|
||||
bench_simdjson_static_reflection(my_struct);
|
||||
}
|
||||
if (matches_filter("simdjson_reuse", filter)) {
|
||||
bench_simdjson_static_reflection_reuse(my_struct);
|
||||
}
|
||||
#if SIMDJSON_STATIC_REFLECTION
|
||||
if (matches_filter("simdjson_to", filter)) {
|
||||
bench_simdjson_to(my_struct);
|
||||
}
|
||||
if (matches_filter("simdjson_to_reuse", filter)) {
|
||||
bench_simdjson_to_reuse(my_struct);
|
||||
}
|
||||
#endif
|
||||
#ifdef SIMDJSON_RUST_VERSION
|
||||
if (matches_filter("rust", filter)) {
|
||||
printf("# WARNING: The Rust benchmark may not be directly comparable since it does not use an equivalent data structure.\n");
|
||||
// Create a Rust-compatible CitmCatalog structure from the JSON string
|
||||
serde_benchmark::CitmCatalog* rust_data =
|
||||
serde_benchmark::citm_from_str(json_str.c_str(), json_str.size());
|
||||
serde_benchmark::citm_from_str(json_str.data(), json_str.size());
|
||||
|
||||
if (rust_data == nullptr) {
|
||||
printf("# Failed to initialize Rust data structure\n");
|
||||
|
||||
@@ -4,79 +4,65 @@
|
||||
#include <string>
|
||||
#include <vector>
|
||||
#include <map>
|
||||
#include <optional>
|
||||
#include <cstdint>
|
||||
|
||||
struct Area {
|
||||
int64_t id;
|
||||
std::string name;
|
||||
int64_t parent;
|
||||
std::vector<int64_t> childAreas;
|
||||
bool operator==(const Area &other) const = default;
|
||||
// Price structure - field names must match JSON keys for reflection
|
||||
struct CITMPrice {
|
||||
uint64_t amount;
|
||||
uint64_t audienceSubCategoryId;
|
||||
uint64_t seatCategoryId;
|
||||
bool operator==(const CITMPrice&) const = default;
|
||||
};
|
||||
|
||||
struct AudienceSubCategory {
|
||||
int64_t id;
|
||||
std::string name;
|
||||
int64_t parent;
|
||||
bool operator==(const AudienceSubCategory &other) const = default;
|
||||
struct CITMArea {
|
||||
uint64_t areaId;
|
||||
std::vector<uint64_t> blockIds;
|
||||
bool operator==(const CITMArea&) const = default;
|
||||
};
|
||||
|
||||
struct Event {
|
||||
int64_t id;
|
||||
std::string name;
|
||||
std::string description;
|
||||
int64_t subTopic;
|
||||
int64_t topic;
|
||||
std::vector<int64_t> audience;
|
||||
bool operator==(const Event &other) const = default;
|
||||
struct CITMSeatCategory {
|
||||
std::vector<CITMArea> areas;
|
||||
uint64_t seatCategoryId;
|
||||
bool operator==(const CITMSeatCategory&) const = default;
|
||||
};
|
||||
|
||||
struct Performance {
|
||||
int64_t id;
|
||||
std::string name;
|
||||
int64_t event;
|
||||
std::string start;
|
||||
int64_t venueCode;
|
||||
bool operator==(const Performance &other) const = default;
|
||||
struct CITMPerformance {
|
||||
uint64_t id;
|
||||
uint64_t eventId;
|
||||
std::optional<std::string> logo;
|
||||
std::optional<std::string> name;
|
||||
std::vector<CITMPrice> prices;
|
||||
std::vector<CITMSeatCategory> seatCategories;
|
||||
std::optional<std::string> seatMapImage;
|
||||
uint64_t start;
|
||||
std::string venueCode;
|
||||
bool operator==(const CITMPerformance&) const = default;
|
||||
};
|
||||
|
||||
struct SeatCategory {
|
||||
int64_t id;
|
||||
std::string name;
|
||||
std::vector<int64_t> areas;
|
||||
bool operator==(const SeatCategory &other) const = default;
|
||||
};
|
||||
|
||||
struct SubTopic {
|
||||
int64_t id;
|
||||
std::string name;
|
||||
int64_t parent;
|
||||
bool operator==(const SubTopic &other) const = default;
|
||||
};
|
||||
|
||||
struct Topic {
|
||||
int64_t id;
|
||||
std::string name;
|
||||
bool operator==(const Topic &other) const = default;
|
||||
};
|
||||
|
||||
struct Venue {
|
||||
int64_t id;
|
||||
std::string name;
|
||||
int64_t address;
|
||||
bool operator==(const Venue &other) const = default;
|
||||
struct CITMEvent {
|
||||
uint64_t id;
|
||||
std::string name;
|
||||
std::optional<std::string> description;
|
||||
std::optional<std::string> logo;
|
||||
std::vector<uint64_t> subTopicIds;
|
||||
std::optional<std::string> subjectCode;
|
||||
std::optional<std::string> subtitle;
|
||||
std::vector<uint64_t> topicIds;
|
||||
bool operator==(const CITMEvent&) const = default;
|
||||
};
|
||||
|
||||
struct CitmCatalog {
|
||||
std::map<std::string, Area> areas;
|
||||
std::map<std::string, AudienceSubCategory> audienceSubCategory;
|
||||
std::map<std::string, Event> events;
|
||||
std::map<std::string, Performance> performances;
|
||||
std::map<std::string, SeatCategory> seatCategory;
|
||||
std::map<std::string, SubTopic> subTopic;
|
||||
std::map<std::string, Topic> topic;
|
||||
std::map<std::string, Venue> venue;
|
||||
|
||||
bool operator==(const CitmCatalog &other) const = default;
|
||||
std::map<std::string, CITMEvent> events;
|
||||
std::vector<CITMPerformance> performances;
|
||||
bool operator==(const CitmCatalog&) const = default;
|
||||
};
|
||||
|
||||
#endif
|
||||
// Type aliases
|
||||
using Event = CITMEvent;
|
||||
using Performance = CITMPerformance;
|
||||
using Price = CITMPrice;
|
||||
using SeatArea = CITMArea;
|
||||
using SeatCategoryInfo = CITMSeatCategory;
|
||||
|
||||
#endif
|
||||
@@ -8,164 +8,72 @@
|
||||
|
||||
using json = nlohmann::json;
|
||||
|
||||
// ---- Area ----
|
||||
inline void to_json(json &j, const Area &a) {
|
||||
// ---- CITMPrice ----
|
||||
inline void to_json(json &j, const CITMPrice &p) {
|
||||
j = json{
|
||||
{"id", a.id},
|
||||
{"name", a.name},
|
||||
{"parent", a.parent},
|
||||
{"childAreas", a.childAreas}
|
||||
{"amount", p.amount},
|
||||
{"audienceSubCategoryId", p.audienceSubCategoryId},
|
||||
{"seatCategoryId", p.seatCategoryId}
|
||||
};
|
||||
}
|
||||
inline void from_json(const json &j, Area &a) {
|
||||
j.at("id").get_to(a.id);
|
||||
j.at("name").get_to(a.name);
|
||||
j.at("parent").get_to(a.parent);
|
||||
j.at("childAreas").get_to(a.childAreas);
|
||||
}
|
||||
|
||||
// ---- AudienceSubCategory ----
|
||||
inline void to_json(json &j, const AudienceSubCategory &asc) {
|
||||
// ---- CITMArea ----
|
||||
inline void to_json(json &j, const CITMArea &a) {
|
||||
j = json{
|
||||
{"id", asc.id},
|
||||
{"name", asc.name},
|
||||
{"parent", asc.parent}
|
||||
{"areaId", a.areaId},
|
||||
{"blockIds", a.blockIds}
|
||||
};
|
||||
}
|
||||
inline void from_json(const json &j, AudienceSubCategory &asc) {
|
||||
j.at("id").get_to(asc.id);
|
||||
j.at("name").get_to(asc.name);
|
||||
j.at("parent").get_to(asc.parent);
|
||||
}
|
||||
|
||||
// ---- Event ----
|
||||
inline void to_json(json &j, const Event &e) {
|
||||
// ---- CITMSeatCategory ----
|
||||
inline void to_json(json &j, const CITMSeatCategory &s) {
|
||||
j = json{
|
||||
{"id", e.id},
|
||||
{"name", e.name},
|
||||
{"description", e.description},
|
||||
{"subTopic", e.subTopic},
|
||||
{"topic", e.topic},
|
||||
{"audience", e.audience}
|
||||
{"areas", s.areas},
|
||||
{"seatCategoryId", s.seatCategoryId}
|
||||
};
|
||||
}
|
||||
inline void from_json(const json &j, Event &e) {
|
||||
j.at("id").get_to(e.id);
|
||||
j.at("name").get_to(e.name);
|
||||
j.at("description").get_to(e.description);
|
||||
j.at("subTopic").get_to(e.subTopic);
|
||||
j.at("topic").get_to(e.topic);
|
||||
j.at("audience").get_to(e.audience);
|
||||
}
|
||||
|
||||
// ---- Performance ----
|
||||
inline void to_json(json &j, const Performance &p) {
|
||||
// ---- CITMPerformance ----
|
||||
inline void to_json(json &j, const CITMPerformance &p) {
|
||||
j = json{
|
||||
{"id", p.id},
|
||||
{"eventId", p.eventId},
|
||||
{"logo", p.logo},
|
||||
{"name", p.name},
|
||||
{"event", p.event},
|
||||
{"prices", p.prices},
|
||||
{"seatCategories", p.seatCategories},
|
||||
{"seatMapImage", p.seatMapImage},
|
||||
{"start", p.start},
|
||||
{"venueCode", p.venueCode}
|
||||
};
|
||||
}
|
||||
inline void from_json(const json &j, Performance &p) {
|
||||
j.at("id").get_to(p.id);
|
||||
j.at("name").get_to(p.name);
|
||||
j.at("event").get_to(p.event);
|
||||
j.at("start").get_to(p.start);
|
||||
j.at("venueCode").get_to(p.venueCode);
|
||||
}
|
||||
|
||||
// ---- SeatCategory ----
|
||||
inline void to_json(json &j, const SeatCategory &sc) {
|
||||
// ---- CITMEvent ----
|
||||
inline void to_json(json &j, const CITMEvent &e) {
|
||||
j = json{
|
||||
{"id", sc.id},
|
||||
{"name", sc.name},
|
||||
{"areas", sc.areas}
|
||||
{"id", e.id},
|
||||
{"name", e.name},
|
||||
{"description", e.description},
|
||||
{"logo", e.logo},
|
||||
{"subTopicIds", e.subTopicIds},
|
||||
{"subjectCode", e.subjectCode},
|
||||
{"subtitle", e.subtitle},
|
||||
{"topicIds", e.topicIds}
|
||||
};
|
||||
}
|
||||
inline void from_json(const json &j, SeatCategory &sc) {
|
||||
j.at("id").get_to(sc.id);
|
||||
j.at("name").get_to(sc.name);
|
||||
j.at("areas").get_to(sc.areas);
|
||||
}
|
||||
|
||||
// ---- SubTopic ----
|
||||
inline void to_json(json &j, const SubTopic &st) {
|
||||
j = json{
|
||||
{"id", st.id},
|
||||
{"name", st.name},
|
||||
{"parent", st.parent}
|
||||
};
|
||||
}
|
||||
inline void from_json(const json &j, SubTopic &st) {
|
||||
j.at("id").get_to(st.id);
|
||||
j.at("name").get_to(st.name);
|
||||
j.at("parent").get_to(st.parent);
|
||||
}
|
||||
|
||||
// ---- Topic ----
|
||||
inline void to_json(json &j, const Topic &t) {
|
||||
j = json{
|
||||
{"id", t.id},
|
||||
{"name", t.name}
|
||||
};
|
||||
}
|
||||
inline void from_json(const json &j, Topic &t) {
|
||||
j.at("id").get_to(t.id);
|
||||
j.at("name").get_to(t.name);
|
||||
}
|
||||
|
||||
// ---- Venue ----
|
||||
inline void to_json(json &j, const Venue &v) {
|
||||
j = json{
|
||||
{"id", v.id},
|
||||
{"name", v.name},
|
||||
{"address", v.address}
|
||||
};
|
||||
}
|
||||
inline void from_json(const json &j, Venue &v) {
|
||||
j.at("id").get_to(v.id);
|
||||
j.at("name").get_to(v.name);
|
||||
j.at("address").get_to(v.address);
|
||||
}
|
||||
|
||||
// ---- CitmCatalog ----
|
||||
inline void to_json(json &j, const CitmCatalog &c) {
|
||||
j = json{
|
||||
{"areas", c.areas},
|
||||
{"audienceSubCategory", c.audienceSubCategory},
|
||||
{"events", c.events},
|
||||
{"performances", c.performances},
|
||||
{"seatCategory", c.seatCategory},
|
||||
{"subTopic", c.subTopic},
|
||||
{"topic", c.topic},
|
||||
{"venue", c.venue}
|
||||
{"performances", c.performances}
|
||||
};
|
||||
}
|
||||
inline void from_json(const json &j, CitmCatalog &c) {
|
||||
j.at("areas").get_to(c.areas);
|
||||
j.at("audienceSubCategory").get_to(c.audienceSubCategory);
|
||||
j.at("events").get_to(c.events);
|
||||
j.at("performances").get_to(c.performances);
|
||||
j.at("seatCategory").get_to(c.seatCategory);
|
||||
j.at("subTopic").get_to(c.subTopic);
|
||||
j.at("topic").get_to(c.topic);
|
||||
j.at("venue").get_to(c.venue);
|
||||
}
|
||||
|
||||
// Optional convenience functions for benchmarking
|
||||
// Serialization function
|
||||
inline std::string nlohmann_serialize(const CitmCatalog &catalog) {
|
||||
json j = catalog;
|
||||
return j.dump();
|
||||
}
|
||||
inline bool nlohmann_deserialize(const std::string &json_in, CitmCatalog &catalog) {
|
||||
try {
|
||||
catalog = json::parse(json_in);
|
||||
return false; // success
|
||||
} catch(...) {
|
||||
return true; // failure
|
||||
}
|
||||
}
|
||||
|
||||
#endif // NLOHMANN_CITM_CATALOG_DATA_H
|
||||
@@ -0,0 +1,189 @@
|
||||
#ifndef RAPIDJSON_CITM_CATALOG_DATA_H
|
||||
#define RAPIDJSON_CITM_CATALOG_DATA_H
|
||||
|
||||
#include "citm_catalog_data.h"
|
||||
#include <rapidjson/document.h>
|
||||
#include <rapidjson/writer.h>
|
||||
#include <rapidjson/stringbuffer.h>
|
||||
#include <rapidjson/error/en.h>
|
||||
|
||||
using namespace rapidjson;
|
||||
|
||||
// RapidJSON deserialization for CITM Catalog data
|
||||
CitmCatalog rapidjson_deserialize_citm(const std::string& json_str) {
|
||||
Document doc;
|
||||
doc.Parse(json_str.c_str());
|
||||
|
||||
if (doc.HasParseError()) {
|
||||
throw std::runtime_error("RapidJSON parse error");
|
||||
}
|
||||
|
||||
CitmCatalog catalog;
|
||||
|
||||
// Parse events
|
||||
if (doc.HasMember("events") && doc["events"].IsObject()) {
|
||||
const Value& events = doc["events"];
|
||||
for (auto it = events.MemberBegin(); it != events.MemberEnd(); ++it) {
|
||||
Event event;
|
||||
const Value& ev = it->value;
|
||||
|
||||
if (ev.HasMember("description") && ev["description"].IsString())
|
||||
event.description = ev["description"].GetString();
|
||||
if (ev.HasMember("id") && ev["id"].IsUint64())
|
||||
event.id = ev["id"].GetUint64();
|
||||
if (ev.HasMember("logo") && ev["logo"].IsString())
|
||||
event.logo = ev["logo"].GetString();
|
||||
if (ev.HasMember("name") && ev["name"].IsString())
|
||||
event.name = ev["name"].GetString();
|
||||
if (ev.HasMember("subjectCode") && ev["subjectCode"].IsString())
|
||||
event.subjectCode = ev["subjectCode"].GetString();
|
||||
if (ev.HasMember("subtitle") && ev["subtitle"].IsString())
|
||||
event.subtitle = ev["subtitle"].GetString();
|
||||
|
||||
if (ev.HasMember("topicIds") && ev["topicIds"].IsArray()) {
|
||||
const Value& topics = ev["topicIds"];
|
||||
for (SizeType j = 0; j < topics.Size(); j++) {
|
||||
if (topics[j].IsUint64())
|
||||
event.topicIds.push_back(topics[j].GetUint64());
|
||||
}
|
||||
}
|
||||
|
||||
if (ev.HasMember("subTopicIds") && ev["subTopicIds"].IsArray()) {
|
||||
const Value& subtopics = ev["subTopicIds"];
|
||||
for (SizeType j = 0; j < subtopics.Size(); j++) {
|
||||
if (subtopics[j].IsUint64())
|
||||
event.subTopicIds.push_back(subtopics[j].GetUint64());
|
||||
}
|
||||
}
|
||||
|
||||
catalog.events[it->name.GetString()] = event;
|
||||
}
|
||||
}
|
||||
|
||||
// Parse performances
|
||||
if (doc.HasMember("performances") && doc["performances"].IsArray()) {
|
||||
const Value& performances = doc["performances"];
|
||||
for (SizeType i = 0; i < performances.Size(); i++) {
|
||||
Performance perf;
|
||||
const Value& p = performances[i];
|
||||
|
||||
if (p.HasMember("id") && p["id"].IsUint64())
|
||||
perf.id = p["id"].GetUint64();
|
||||
if (p.HasMember("eventId") && p["eventId"].IsUint64())
|
||||
perf.eventId = p["eventId"].GetUint64();
|
||||
if (p.HasMember("start") && p["start"].IsUint64())
|
||||
perf.start = p["start"].GetUint64();
|
||||
if (p.HasMember("venueCode") && p["venueCode"].IsString())
|
||||
perf.venueCode = p["venueCode"].GetString();
|
||||
if (p.HasMember("name") && p["name"].IsString())
|
||||
perf.name = p["name"].GetString();
|
||||
|
||||
catalog.performances.push_back(perf);
|
||||
}
|
||||
}
|
||||
|
||||
// Parse other string maps
|
||||
auto parseStringMap = [&doc](const char* key, std::map<std::string, std::string>& target) {
|
||||
if (doc.HasMember(key) && doc[key].IsObject()) {
|
||||
const Value& obj = doc[key];
|
||||
for (auto it = obj.MemberBegin(); it != obj.MemberEnd(); ++it) {
|
||||
if (it->value.IsString()) {
|
||||
target[it->name.GetString()] = it->value.GetString();
|
||||
}
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
return catalog;
|
||||
}
|
||||
|
||||
// RapidJSON serialization for CITM Catalog data
|
||||
std::string rapidjson_serialize_citm(const CitmCatalog& catalog) {
|
||||
Document doc;
|
||||
doc.SetObject();
|
||||
Document::AllocatorType& allocator = doc.GetAllocator();
|
||||
|
||||
// Serialize events
|
||||
Value events_obj(kObjectType);
|
||||
for (const auto& [key, event] : catalog.events) {
|
||||
Value event_obj(kObjectType);
|
||||
|
||||
if (event.description) {
|
||||
Value desc;
|
||||
desc.SetString(event.description->c_str(), allocator);
|
||||
event_obj.AddMember("description", desc, allocator);
|
||||
}
|
||||
|
||||
event_obj.AddMember("id", event.id, allocator);
|
||||
|
||||
if (event.logo) {
|
||||
Value logo;
|
||||
logo.SetString(event.logo->c_str(), allocator);
|
||||
event_obj.AddMember("logo", logo, allocator);
|
||||
}
|
||||
|
||||
Value name;
|
||||
name.SetString(event.name.c_str(), allocator);
|
||||
event_obj.AddMember("name", name, allocator);
|
||||
|
||||
if (event.subjectCode) {
|
||||
Value subject;
|
||||
subject.SetString(event.subjectCode->c_str(), allocator);
|
||||
event_obj.AddMember("subjectCode", subject, allocator);
|
||||
}
|
||||
|
||||
if (event.subtitle) {
|
||||
Value subtitle;
|
||||
subtitle.SetString(event.subtitle->c_str(), allocator);
|
||||
event_obj.AddMember("subtitle", subtitle, allocator);
|
||||
}
|
||||
|
||||
Value topicIds(kArrayType);
|
||||
for (uint64_t id : event.topicIds) {
|
||||
topicIds.PushBack(id, allocator);
|
||||
}
|
||||
event_obj.AddMember("topicIds", topicIds, allocator);
|
||||
|
||||
Value subTopicIds(kArrayType);
|
||||
for (uint64_t id : event.subTopicIds) {
|
||||
subTopicIds.PushBack(id, allocator);
|
||||
}
|
||||
event_obj.AddMember("subTopicIds", subTopicIds, allocator);
|
||||
|
||||
Value key_val;
|
||||
key_val.SetString(key.c_str(), allocator);
|
||||
events_obj.AddMember(key_val, event_obj, allocator);
|
||||
}
|
||||
doc.AddMember("events", events_obj, allocator);
|
||||
|
||||
// Serialize performances
|
||||
Value performances_array(kArrayType);
|
||||
for (const auto& perf : catalog.performances) {
|
||||
Value perf_obj(kObjectType);
|
||||
perf_obj.AddMember("id", perf.id, allocator);
|
||||
perf_obj.AddMember("eventId", perf.eventId, allocator);
|
||||
perf_obj.AddMember("start", perf.start, allocator);
|
||||
|
||||
Value venue;
|
||||
venue.SetString(perf.venueCode.c_str(), allocator);
|
||||
perf_obj.AddMember("venueCode", venue, allocator);
|
||||
|
||||
if (perf.name) {
|
||||
Value name;
|
||||
name.SetString(perf.name->c_str(), allocator);
|
||||
perf_obj.AddMember("name", name, allocator);
|
||||
}
|
||||
|
||||
|
||||
performances_array.PushBack(perf_obj, allocator);
|
||||
}
|
||||
doc.AddMember("performances", performances_array, allocator);
|
||||
|
||||
StringBuffer buffer;
|
||||
Writer<StringBuffer> writer(buffer);
|
||||
doc.Accept(writer);
|
||||
|
||||
return buffer.GetString();
|
||||
}
|
||||
|
||||
#endif // RAPIDJSON_CITM_CATALOG_DATA_H
|
||||
@@ -0,0 +1,231 @@
|
||||
#ifndef YYJSON_CITM_CATALOG_DATA_H
|
||||
#define YYJSON_CITM_CATALOG_DATA_H
|
||||
|
||||
#include "citm_catalog_data.h"
|
||||
#include <yyjson.h>
|
||||
#include <string>
|
||||
#include <stdexcept>
|
||||
|
||||
// yyjson deserialization for CITM Catalog data
|
||||
// Matches C++ CitmCatalog struct (only events + performances)
|
||||
CitmCatalog yyjson_deserialize_citm(const std::string &json_str) {
|
||||
CitmCatalog catalog;
|
||||
|
||||
yyjson_doc *doc = yyjson_read(json_str.c_str(), json_str.size(), 0);
|
||||
if (!doc) {
|
||||
throw std::runtime_error("yyjson parse error");
|
||||
}
|
||||
|
||||
yyjson_val *root = yyjson_doc_get_root(doc);
|
||||
if (!root) {
|
||||
yyjson_doc_free(doc);
|
||||
return catalog;
|
||||
}
|
||||
|
||||
// Parse events
|
||||
yyjson_val *events_val = yyjson_obj_get(root, "events");
|
||||
if (events_val && yyjson_is_obj(events_val)) {
|
||||
size_t idx, max;
|
||||
yyjson_val *key, *val;
|
||||
yyjson_obj_foreach(events_val, idx, max, key, val) {
|
||||
CITMEvent event;
|
||||
|
||||
yyjson_val *v;
|
||||
v = yyjson_obj_get(val, "description");
|
||||
if (v && yyjson_is_str(v)) event.description = yyjson_get_str(v);
|
||||
|
||||
v = yyjson_obj_get(val, "id");
|
||||
if (v && yyjson_is_uint(v)) event.id = yyjson_get_uint(v);
|
||||
|
||||
v = yyjson_obj_get(val, "logo");
|
||||
if (v && yyjson_is_str(v)) event.logo = yyjson_get_str(v);
|
||||
|
||||
v = yyjson_obj_get(val, "name");
|
||||
if (v && yyjson_is_str(v)) event.name = yyjson_get_str(v);
|
||||
|
||||
v = yyjson_obj_get(val, "subjectCode");
|
||||
if (v && yyjson_is_str(v)) event.subjectCode = yyjson_get_str(v);
|
||||
|
||||
v = yyjson_obj_get(val, "subtitle");
|
||||
if (v && yyjson_is_str(v)) event.subtitle = yyjson_get_str(v);
|
||||
|
||||
// Parse topicIds array
|
||||
v = yyjson_obj_get(val, "topicIds");
|
||||
if (v && yyjson_is_arr(v)) {
|
||||
size_t arr_idx, arr_max;
|
||||
yyjson_val *arr_val;
|
||||
yyjson_arr_foreach(v, arr_idx, arr_max, arr_val) {
|
||||
if (yyjson_is_uint(arr_val))
|
||||
event.topicIds.push_back(yyjson_get_uint(arr_val));
|
||||
}
|
||||
}
|
||||
|
||||
// Parse subTopicIds array
|
||||
v = yyjson_obj_get(val, "subTopicIds");
|
||||
if (v && yyjson_is_arr(v)) {
|
||||
size_t arr_idx, arr_max;
|
||||
yyjson_val *arr_val;
|
||||
yyjson_arr_foreach(v, arr_idx, arr_max, arr_val) {
|
||||
if (yyjson_is_uint(arr_val))
|
||||
event.subTopicIds.push_back(yyjson_get_uint(arr_val));
|
||||
}
|
||||
}
|
||||
|
||||
if (yyjson_is_str(key))
|
||||
catalog.events[yyjson_get_str(key)] = event;
|
||||
}
|
||||
}
|
||||
|
||||
// Parse performances (simplified - full parsing would need prices/seatCategories)
|
||||
yyjson_val *performances_val = yyjson_obj_get(root, "performances");
|
||||
if (performances_val && yyjson_is_arr(performances_val)) {
|
||||
size_t idx, max;
|
||||
yyjson_val *perf_val;
|
||||
yyjson_arr_foreach(performances_val, idx, max, perf_val) {
|
||||
CITMPerformance perf;
|
||||
|
||||
yyjson_val *v;
|
||||
v = yyjson_obj_get(perf_val, "id");
|
||||
if (v && yyjson_is_uint(v)) perf.id = yyjson_get_uint(v);
|
||||
|
||||
v = yyjson_obj_get(perf_val, "eventId");
|
||||
if (v && yyjson_is_uint(v)) perf.eventId = yyjson_get_uint(v);
|
||||
|
||||
v = yyjson_obj_get(perf_val, "start");
|
||||
if (v && yyjson_is_uint(v)) perf.start = yyjson_get_uint(v);
|
||||
|
||||
v = yyjson_obj_get(perf_val, "venueCode");
|
||||
if (v && yyjson_is_str(v)) perf.venueCode = yyjson_get_str(v);
|
||||
|
||||
v = yyjson_obj_get(perf_val, "name");
|
||||
if (v && yyjson_is_str(v)) perf.name = yyjson_get_str(v);
|
||||
|
||||
v = yyjson_obj_get(perf_val, "logo");
|
||||
if (v && yyjson_is_str(v)) perf.logo = yyjson_get_str(v);
|
||||
|
||||
v = yyjson_obj_get(perf_val, "seatMapImage");
|
||||
if (v && yyjson_is_str(v)) perf.seatMapImage = yyjson_get_str(v);
|
||||
|
||||
// Note: prices and seatCategories parsing omitted for brevity
|
||||
// The serialization benchmark uses data loaded by simdjson
|
||||
|
||||
catalog.performances.push_back(perf);
|
||||
}
|
||||
}
|
||||
|
||||
yyjson_doc_free(doc);
|
||||
return catalog;
|
||||
}
|
||||
|
||||
// Helper to add optional string field
|
||||
static inline void yyjson_add_optional_str(yyjson_mut_doc *doc, yyjson_mut_val *obj,
|
||||
const char *key, const std::optional<std::string> &val) {
|
||||
if (val.has_value()) {
|
||||
yyjson_mut_obj_add_str(doc, obj, key, val->c_str());
|
||||
} else {
|
||||
yyjson_mut_obj_add_null(doc, obj, key);
|
||||
}
|
||||
}
|
||||
|
||||
// yyjson serialization for CITM Catalog data
|
||||
// Matches C++ CitmCatalog struct exactly (only events + performances)
|
||||
std::string yyjson_serialize_citm(const CitmCatalog &catalog) {
|
||||
yyjson_mut_doc *doc = yyjson_mut_doc_new(NULL);
|
||||
yyjson_mut_val *root = yyjson_mut_obj(doc);
|
||||
yyjson_mut_doc_set_root(doc, root);
|
||||
|
||||
// Create events object
|
||||
yyjson_mut_val *events_obj = yyjson_mut_obj(doc);
|
||||
for (const auto& [key, event] : catalog.events) {
|
||||
yyjson_mut_val *event_obj = yyjson_mut_obj(doc);
|
||||
|
||||
yyjson_add_optional_str(doc, event_obj, "description", event.description);
|
||||
yyjson_mut_obj_add_uint(doc, event_obj, "id", event.id);
|
||||
yyjson_add_optional_str(doc, event_obj, "logo", event.logo);
|
||||
// name is not optional in CITMEvent
|
||||
yyjson_mut_obj_add_str(doc, event_obj, "name", event.name.c_str());
|
||||
|
||||
// Add subTopicIds array
|
||||
yyjson_mut_val *subtopic_ids = yyjson_mut_arr(doc);
|
||||
for (uint64_t id : event.subTopicIds) {
|
||||
yyjson_mut_arr_add_uint(doc, subtopic_ids, id);
|
||||
}
|
||||
yyjson_mut_obj_add_val(doc, event_obj, "subTopicIds", subtopic_ids);
|
||||
|
||||
yyjson_add_optional_str(doc, event_obj, "subjectCode", event.subjectCode);
|
||||
yyjson_add_optional_str(doc, event_obj, "subtitle", event.subtitle);
|
||||
|
||||
// Add topicIds array
|
||||
yyjson_mut_val *topic_ids = yyjson_mut_arr(doc);
|
||||
for (uint64_t id : event.topicIds) {
|
||||
yyjson_mut_arr_add_uint(doc, topic_ids, id);
|
||||
}
|
||||
yyjson_mut_obj_add_val(doc, event_obj, "topicIds", topic_ids);
|
||||
|
||||
yyjson_mut_obj_add_val(doc, events_obj, key.c_str(), event_obj);
|
||||
}
|
||||
yyjson_mut_obj_add_val(doc, root, "events", events_obj);
|
||||
|
||||
// Create performances array
|
||||
yyjson_mut_val *performances_array = yyjson_mut_arr(doc);
|
||||
for (const auto& perf : catalog.performances) {
|
||||
yyjson_mut_val *perf_obj = yyjson_mut_obj(doc);
|
||||
|
||||
yyjson_mut_obj_add_uint(doc, perf_obj, "eventId", perf.eventId);
|
||||
yyjson_mut_obj_add_uint(doc, perf_obj, "id", perf.id);
|
||||
yyjson_add_optional_str(doc, perf_obj, "logo", perf.logo);
|
||||
yyjson_add_optional_str(doc, perf_obj, "name", perf.name);
|
||||
|
||||
// Add prices array
|
||||
yyjson_mut_val *prices_array = yyjson_mut_arr(doc);
|
||||
for (const auto& price : perf.prices) {
|
||||
yyjson_mut_val *price_obj = yyjson_mut_obj(doc);
|
||||
yyjson_mut_obj_add_uint(doc, price_obj, "amount", price.amount);
|
||||
yyjson_mut_obj_add_uint(doc, price_obj, "audienceSubCategoryId", price.audienceSubCategoryId);
|
||||
yyjson_mut_obj_add_uint(doc, price_obj, "seatCategoryId", price.seatCategoryId);
|
||||
yyjson_mut_arr_append(prices_array, price_obj);
|
||||
}
|
||||
yyjson_mut_obj_add_val(doc, perf_obj, "prices", prices_array);
|
||||
|
||||
// Add seatCategories array
|
||||
yyjson_mut_val *seat_cats_array = yyjson_mut_arr(doc);
|
||||
for (const auto& seatCat : perf.seatCategories) {
|
||||
yyjson_mut_val *seat_cat_obj = yyjson_mut_obj(doc);
|
||||
|
||||
// Add areas array
|
||||
yyjson_mut_val *areas_array = yyjson_mut_arr(doc);
|
||||
for (const auto& area : seatCat.areas) {
|
||||
yyjson_mut_val *area_obj = yyjson_mut_obj(doc);
|
||||
yyjson_mut_obj_add_uint(doc, area_obj, "areaId", area.areaId);
|
||||
|
||||
yyjson_mut_val *block_ids = yyjson_mut_arr(doc);
|
||||
for (uint64_t blockId : area.blockIds) {
|
||||
yyjson_mut_arr_add_uint(doc, block_ids, blockId);
|
||||
}
|
||||
yyjson_mut_obj_add_val(doc, area_obj, "blockIds", block_ids);
|
||||
yyjson_mut_arr_append(areas_array, area_obj);
|
||||
}
|
||||
yyjson_mut_obj_add_val(doc, seat_cat_obj, "areas", areas_array);
|
||||
yyjson_mut_obj_add_uint(doc, seat_cat_obj, "seatCategoryId", seatCat.seatCategoryId);
|
||||
yyjson_mut_arr_append(seat_cats_array, seat_cat_obj);
|
||||
}
|
||||
yyjson_mut_obj_add_val(doc, perf_obj, "seatCategories", seat_cats_array);
|
||||
|
||||
yyjson_add_optional_str(doc, perf_obj, "seatMapImage", perf.seatMapImage);
|
||||
yyjson_mut_obj_add_uint(doc, perf_obj, "start", perf.start);
|
||||
yyjson_mut_obj_add_str(doc, perf_obj, "venueCode", perf.venueCode.c_str());
|
||||
|
||||
yyjson_mut_arr_append(performances_array, perf_obj);
|
||||
}
|
||||
yyjson_mut_obj_add_val(doc, root, "performances", performances_array);
|
||||
|
||||
// Write to string
|
||||
char *json_output = yyjson_mut_write(doc, 0, NULL);
|
||||
std::string result(json_output);
|
||||
free(json_output);
|
||||
yyjson_mut_doc_free(doc);
|
||||
|
||||
return result;
|
||||
}
|
||||
|
||||
#endif // YYJSON_CITM_CATALOG_DATA_H
|
||||
@@ -5,405 +5,165 @@ extern crate libc;
|
||||
use libc::{c_char, size_t};
|
||||
use serde::{Serialize, Deserialize};
|
||||
use std::{collections::HashMap, ffi::CString, ptr, slice};
|
||||
use serde::de::{self, Deserializer};
|
||||
/******************************************************/
|
||||
/******************************************************/
|
||||
/**
|
||||
* Warning: the C++ code may not generate the same JSON.
|
||||
*/
|
||||
/******************************************************/
|
||||
/******************************************************/
|
||||
|
||||
// This has no equivalent in C++:
|
||||
#[derive(Serialize, Deserialize)]
|
||||
pub struct Metadata {
|
||||
result_type: String,
|
||||
iso_language_code: String,
|
||||
}
|
||||
//==============================================================================
|
||||
// Twitter Benchmark Structures
|
||||
// These match the C++ TwitterData structures exactly
|
||||
//==============================================================================
|
||||
|
||||
#[derive(Serialize, Deserialize)]
|
||||
pub struct User {
|
||||
id: i64,
|
||||
id_str: String,
|
||||
id: u64,
|
||||
name: String,
|
||||
screen_name: String,
|
||||
location: String,
|
||||
description: String,
|
||||
// C++ does not have those:
|
||||
// url: Option<String>,
|
||||
//protected: bool,
|
||||
//listed_count: i64,
|
||||
//created_at: String,
|
||||
//favourites_count: i64,
|
||||
//utc_offset: Option<i64>,
|
||||
//time_zone: Option<String>,
|
||||
//geo_enabled: bool,
|
||||
verified: bool,
|
||||
followers_count: i64,
|
||||
friends_count: i64,
|
||||
statuses_count: i64,
|
||||
// C++ does not have those:
|
||||
//lang: String,
|
||||
//profile_background_color: String,
|
||||
//profile_background_image_url: String,
|
||||
//profile_background_image_url_https: String,
|
||||
//profile_background_tile: bool,
|
||||
//profile_image_url: String,
|
||||
//profile_image_url_https: String,
|
||||
//profile_banner_url: Option<String>,
|
||||
//profile_link_color: String,
|
||||
//profile_sidebar_border_color: String,
|
||||
//profile_sidebar_fill_color: String,
|
||||
//profile_text_color: String,
|
||||
//profile_use_background_image: bool,
|
||||
//default_profile: bool,
|
||||
//default_profile_image: bool,
|
||||
//following: bool,
|
||||
//follow_request_sent: bool,
|
||||
//notifications: bool,
|
||||
}
|
||||
|
||||
#[derive(Serialize, Deserialize)]
|
||||
pub struct Hashtag {
|
||||
text: String,
|
||||
|
||||
// C++ has those but D. Lemire does not know what they are, they don't appear in the JSON:
|
||||
// int64_t indices_start;
|
||||
// int64_t indices_end;
|
||||
}
|
||||
|
||||
#[derive(Serialize, Deserialize)]
|
||||
pub struct Url {
|
||||
url: String,
|
||||
expanded_url: String,
|
||||
display_url: String,
|
||||
// C++ has those but D. Lemire does not know what they are, they don't appear in the JSON:
|
||||
// int64_t indices_start;
|
||||
// int64_t indices_end;
|
||||
}
|
||||
|
||||
#[derive(Serialize, Deserialize)]
|
||||
pub struct UserMention {
|
||||
id: i64,
|
||||
name: String,
|
||||
screen_name: String,
|
||||
// Not in the C++ equivalent:
|
||||
//id_str: String,
|
||||
//indices: Vec<i64>,
|
||||
// C++ has those but D. Lemire does not know what they are, they don't appear in the JSON:
|
||||
// int64_t indices_start;
|
||||
// int64_t indices_end;
|
||||
}
|
||||
|
||||
#[derive(Serialize, Deserialize)]
|
||||
pub struct Entities {
|
||||
hashtags: Vec<Hashtag>,
|
||||
urls: Vec<Url>,
|
||||
user_mentions: Vec<UserMention>,
|
||||
followers_count: u64,
|
||||
friends_count: u64,
|
||||
statuses_count: u64,
|
||||
}
|
||||
|
||||
#[derive(Serialize, Deserialize)]
|
||||
pub struct Status {
|
||||
created_at: String,
|
||||
id: i64,
|
||||
id: u64,
|
||||
text: String,
|
||||
user: User,
|
||||
entities: Entities,
|
||||
retweet_count: i64,
|
||||
favorite_count: i64,
|
||||
favorited: bool,
|
||||
retweeted: bool,
|
||||
// None of these are in the C++ equivalent:
|
||||
/*
|
||||
metadata: Metadata,
|
||||
id_str: String,
|
||||
source: String,
|
||||
truncated: bool,
|
||||
in_reply_to_status_id: Option<i64>,
|
||||
in_reply_to_status_id_str: Option<String>,
|
||||
in_reply_to_user_id: Option<i64>,
|
||||
in_reply_to_user_id_str: Option<String>,
|
||||
in_reply_to_screen_name: Option<String>,
|
||||
geo: Option<String>,
|
||||
coordinates: Option<String>,
|
||||
place: Option<String>,
|
||||
contributors: Option<String>,
|
||||
lang: String,
|
||||
*/
|
||||
retweet_count: u64,
|
||||
favorite_count: u64,
|
||||
}
|
||||
|
||||
#[derive(Serialize, Deserialize)]
|
||||
pub struct TwitterData {
|
||||
statuses: Vec<Status>,
|
||||
}
|
||||
static mut TWITTER_DATA: *mut TwitterData = std::ptr::null_mut();
|
||||
|
||||
#[no_mangle]
|
||||
pub unsafe extern "C" fn twitter_from_str(raw_input: *const c_char, raw_input_length: size_t) -> *mut TwitterData {
|
||||
let input = std::str::from_utf8_unchecked(slice::from_raw_parts(raw_input as *const u8, raw_input_length));
|
||||
match serde_json::from_str(&input) {
|
||||
Ok(result) => Box::into_raw(Box::new(result)),
|
||||
Err(_) => std::ptr::null_mut(),
|
||||
}
|
||||
pub unsafe extern "C" fn twitter_from_str(raw_input: *const c_char, raw_input_length: size_t) -> *mut TwitterData {
|
||||
let input = std::str::from_utf8_unchecked(slice::from_raw_parts(raw_input as *const u8, raw_input_length));
|
||||
match serde_json::from_str(&input) {
|
||||
Ok(result) => Box::into_raw(Box::new(result)),
|
||||
Err(_) => std::ptr::null_mut(),
|
||||
}
|
||||
}
|
||||
|
||||
#[no_mangle]
|
||||
pub unsafe extern "C" fn str_from_twitter(raw: *mut TwitterData) -> *const c_char {
|
||||
let twitter_thing = { &*raw };
|
||||
let serialized = serde_json::to_string(&twitter_thing).unwrap();
|
||||
return std::ffi::CString::new(serialized.as_str()).unwrap().into_raw()
|
||||
pub unsafe extern "C" fn set_twitter_data(raw: *mut TwitterData) {
|
||||
TWITTER_DATA = raw;
|
||||
}
|
||||
|
||||
#[no_mangle]
|
||||
pub unsafe extern "C" fn serialize_twitter_to_string() -> usize {
|
||||
if TWITTER_DATA.is_null() {
|
||||
return 0;
|
||||
}
|
||||
let data = &*TWITTER_DATA;
|
||||
serde_json::to_string(data).unwrap().len()
|
||||
}
|
||||
|
||||
#[no_mangle]
|
||||
pub unsafe extern "C" fn free_twitter(raw: *mut TwitterData) {
|
||||
if raw.is_null() {
|
||||
return;
|
||||
}
|
||||
|
||||
drop(Box::from_raw(raw))
|
||||
if raw.is_null() {
|
||||
return;
|
||||
}
|
||||
drop(Box::from_raw(raw))
|
||||
}
|
||||
|
||||
|
||||
#[no_mangle]
|
||||
pub unsafe extern fn free_string(ptr: *const c_char) {
|
||||
let _ = std::ffi::CString::from_raw(ptr as *mut _);
|
||||
}
|
||||
|
||||
// Functions associated with the CitmCatalog benchmark
|
||||
//==============================================================================
|
||||
// CITM Catalog Benchmark Structures
|
||||
// These match the C++ CitmCatalog structures EXACTLY for fair comparison
|
||||
//==============================================================================
|
||||
|
||||
#[derive(Serialize, Deserialize)]
|
||||
pub struct Area {
|
||||
pub id: i64,
|
||||
pub name: Option<String>, // Changed to Option
|
||||
pub parent: i64,
|
||||
#[serde(rename = "childAreas")]
|
||||
pub child_areas: Vec<i64>,
|
||||
/// Matches C++ CITMPrice struct exactly
|
||||
#[derive(Serialize, Deserialize, Debug, Clone)]
|
||||
pub struct CITMPrice {
|
||||
pub amount: u64,
|
||||
#[serde(rename = "audienceSubCategoryId")]
|
||||
pub audience_sub_category_id: u64,
|
||||
#[serde(rename = "seatCategoryId")]
|
||||
pub seat_category_id: u64,
|
||||
}
|
||||
|
||||
#[derive(Serialize, Deserialize)]
|
||||
pub struct AudienceSubCategory {
|
||||
pub id: i64,
|
||||
pub name: Option<String>, // Changed to Option
|
||||
pub parent: i64,
|
||||
/// Matches C++ CITMArea struct exactly
|
||||
#[derive(Serialize, Deserialize, Debug, Clone)]
|
||||
pub struct CITMArea {
|
||||
#[serde(rename = "areaId")]
|
||||
pub area_id: u64,
|
||||
#[serde(rename = "blockIds")]
|
||||
pub block_ids: Vec<u64>,
|
||||
}
|
||||
|
||||
#[derive(Serialize, Deserialize, Debug)]
|
||||
pub struct Event {
|
||||
#[serde(default)]
|
||||
pub description: Option<String>,
|
||||
pub id: i64,
|
||||
/// Matches C++ CITMSeatCategory struct exactly
|
||||
#[derive(Serialize, Deserialize, Debug, Clone)]
|
||||
pub struct CITMSeatCategory {
|
||||
pub areas: Vec<CITMArea>,
|
||||
#[serde(rename = "seatCategoryId")]
|
||||
pub seat_category_id: u64,
|
||||
}
|
||||
|
||||
/// Matches C++ CITMPerformance struct exactly
|
||||
#[derive(Serialize, Deserialize, Debug, Clone)]
|
||||
pub struct CITMPerformance {
|
||||
pub id: u64,
|
||||
#[serde(rename = "eventId")]
|
||||
pub event_id: u64,
|
||||
#[serde(default)]
|
||||
pub logo: Option<String>,
|
||||
#[serde(default)]
|
||||
pub name: Option<String>,
|
||||
pub prices: Vec<CITMPrice>,
|
||||
#[serde(rename = "seatCategories")]
|
||||
pub seat_categories: Vec<CITMSeatCategory>,
|
||||
#[serde(default)]
|
||||
pub subTopicIds: Vec<i64>,
|
||||
#[serde(rename = "seatMapImage")]
|
||||
pub seat_map_image: Option<String>,
|
||||
pub start: u64,
|
||||
#[serde(rename = "venueCode")]
|
||||
pub venue_code: String,
|
||||
}
|
||||
|
||||
/// Matches C++ CITMEvent struct exactly
|
||||
#[derive(Serialize, Deserialize, Debug, Clone)]
|
||||
pub struct CITMEvent {
|
||||
pub id: u64,
|
||||
#[serde(default)]
|
||||
pub subjectCode: Option<String>,
|
||||
pub name: Option<String>,
|
||||
#[serde(default)]
|
||||
pub description: Option<String>,
|
||||
#[serde(default)]
|
||||
pub logo: Option<String>,
|
||||
#[serde(default)]
|
||||
#[serde(rename = "subTopicIds")]
|
||||
pub sub_topic_ids: Vec<u64>,
|
||||
#[serde(default)]
|
||||
#[serde(rename = "subjectCode")]
|
||||
pub subject_code: Option<String>,
|
||||
#[serde(default)]
|
||||
pub subtitle: Option<String>,
|
||||
#[serde(default)]
|
||||
pub topicIds: Vec<i64>,
|
||||
// Add a catch-all for any other fields
|
||||
#[serde(flatten)]
|
||||
pub extra: HashMap<String, serde_json::Value>,
|
||||
}
|
||||
|
||||
#[derive(Serialize, Deserialize, Debug)]
|
||||
pub struct Performance {
|
||||
#[serde(default)]
|
||||
pub id: i64,
|
||||
|
||||
#[serde(default)]
|
||||
pub name: Option<String>,
|
||||
|
||||
#[serde(default)]
|
||||
pub event: i64,
|
||||
|
||||
// This is the key fix - accept any JSON value type for timestamps
|
||||
// This allows both string dates and integer timestamps (line 3511)
|
||||
#[serde(default)]
|
||||
pub start: serde_json::Value,
|
||||
|
||||
#[serde(rename = "venueCode")]
|
||||
pub venue_code: String,
|
||||
|
||||
// Add a catch-all for any other fields
|
||||
#[serde(flatten)]
|
||||
pub extra: HashMap<String, serde_json::Value>,
|
||||
}
|
||||
|
||||
#[derive(Serialize, Deserialize)]
|
||||
pub struct SeatCategory {
|
||||
pub id: i64,
|
||||
pub name: Option<String>, // Changed to Option
|
||||
pub areas: Vec<i64>,
|
||||
}
|
||||
|
||||
#[derive(Serialize, Deserialize)]
|
||||
pub struct SubTopic {
|
||||
pub id: i64,
|
||||
pub name: Option<String>, // Changed to Option
|
||||
pub parent: i64,
|
||||
}
|
||||
|
||||
#[derive(Serialize, Deserialize)]
|
||||
pub struct Topic {
|
||||
pub id: i64,
|
||||
pub name: Option<String>, // Changed to Option
|
||||
}
|
||||
|
||||
#[derive(Serialize, Deserialize)]
|
||||
pub struct Venue {
|
||||
pub id: i64,
|
||||
pub name: Option<String>, // Changed to Option
|
||||
pub address: i64,
|
||||
}
|
||||
|
||||
// Custom deserializers
|
||||
fn deserialize_string_to_area<'de, D>(deserializer: D) -> Result<HashMap<String, Area>, D::Error>
|
||||
where D: Deserializer<'de> {
|
||||
let string_map: HashMap<String, String> = HashMap::deserialize(deserializer)?;
|
||||
let mut result = HashMap::new();
|
||||
|
||||
for (id, name) in string_map {
|
||||
let id_num = id.parse::<i64>().unwrap_or(0);
|
||||
result.insert(id.clone(), Area {
|
||||
id: id_num,
|
||||
name: Some(name),
|
||||
parent: 0,
|
||||
child_areas: Vec::new(),
|
||||
});
|
||||
}
|
||||
|
||||
Ok(result)
|
||||
}
|
||||
|
||||
fn deserialize_string_to_audience_subcategory<'de, D>(deserializer: D) -> Result<HashMap<String, AudienceSubCategory>, D::Error>
|
||||
where D: Deserializer<'de> {
|
||||
let string_map: HashMap<String, String> = HashMap::deserialize(deserializer)?;
|
||||
let mut result = HashMap::new();
|
||||
|
||||
for (id, name) in string_map {
|
||||
let id_num = id.parse::<i64>().unwrap_or(0);
|
||||
result.insert(id.clone(), AudienceSubCategory {
|
||||
id: id_num,
|
||||
name: Some(name),
|
||||
parent: 0,
|
||||
});
|
||||
}
|
||||
|
||||
Ok(result)
|
||||
}
|
||||
|
||||
fn deserialize_string_to_seat_category<'de, D>(deserializer: D) -> Result<HashMap<String, SeatCategory>, D::Error>
|
||||
where D: Deserializer<'de> {
|
||||
let string_map: HashMap<String, String> = HashMap::deserialize(deserializer)?;
|
||||
let mut result = HashMap::new();
|
||||
|
||||
for (id, name) in string_map {
|
||||
let id_num = id.parse::<i64>().unwrap_or(0);
|
||||
result.insert(id.clone(), SeatCategory {
|
||||
id: id_num,
|
||||
name: Some(name),
|
||||
areas: Vec::new(),
|
||||
});
|
||||
}
|
||||
|
||||
Ok(result)
|
||||
}
|
||||
|
||||
fn deserialize_string_to_subtopic<'de, D>(deserializer: D) -> Result<HashMap<String, SubTopic>, D::Error>
|
||||
where D: Deserializer<'de> {
|
||||
let string_map: HashMap<String, String> = HashMap::deserialize(deserializer)?;
|
||||
let mut result = HashMap::new();
|
||||
|
||||
for (id, name) in string_map {
|
||||
let id_num = id.parse::<i64>().unwrap_or(0);
|
||||
result.insert(id.clone(), SubTopic {
|
||||
id: id_num,
|
||||
name: Some(name),
|
||||
parent: 0,
|
||||
});
|
||||
}
|
||||
|
||||
Ok(result)
|
||||
}
|
||||
|
||||
fn deserialize_string_to_topic<'de, D>(deserializer: D) -> Result<HashMap<String, Topic>, D::Error>
|
||||
where D: Deserializer<'de> {
|
||||
let string_map: HashMap<String, String> = HashMap::deserialize(deserializer)?;
|
||||
let mut result = HashMap::new();
|
||||
|
||||
for (id, name) in string_map {
|
||||
let id_num = id.parse::<i64>().unwrap_or(0);
|
||||
result.insert(id.clone(), Topic {
|
||||
id: id_num,
|
||||
name: Some(name),
|
||||
});
|
||||
}
|
||||
|
||||
Ok(result)
|
||||
}
|
||||
|
||||
fn deserialize_string_to_venue<'de, D>(deserializer: D) -> Result<HashMap<String, Venue>, D::Error>
|
||||
where D: Deserializer<'de> {
|
||||
let string_map: HashMap<String, String> = HashMap::deserialize(deserializer)?;
|
||||
let mut result = HashMap::new();
|
||||
|
||||
for (id, name) in string_map {
|
||||
result.insert(id.clone(), Venue {
|
||||
id: 0,
|
||||
name: Some(name),
|
||||
address: 0,
|
||||
});
|
||||
}
|
||||
|
||||
Ok(result)
|
||||
#[serde(rename = "topicIds")]
|
||||
pub topic_ids: Vec<u64>,
|
||||
}
|
||||
|
||||
/// Matches C++ CitmCatalog struct exactly - ONLY events and performances
|
||||
/// This is the key fix: we serialize only what C++ serializes
|
||||
#[derive(Serialize, Deserialize, Debug)]
|
||||
pub struct CitmCatalog {
|
||||
#[serde(rename = "areaNames")]
|
||||
pub area_names: HashMap<String, String>,
|
||||
|
||||
#[serde(rename = "audienceSubCategoryNames")]
|
||||
pub audience_subcategory_names: HashMap<String, String>,
|
||||
|
||||
#[serde(default)]
|
||||
#[serde(rename = "blockNames")]
|
||||
pub block_names: HashMap<String, String>,
|
||||
|
||||
pub events: HashMap<String, Event>,
|
||||
|
||||
#[serde(default)]
|
||||
pub performances: Vec<Performance>,
|
||||
|
||||
#[serde(rename = "seatCategoryNames")]
|
||||
pub seat_category_names: HashMap<String, String>,
|
||||
|
||||
#[serde(rename = "subTopicNames")]
|
||||
pub subtopic_names: HashMap<String, String>,
|
||||
|
||||
#[serde(default)]
|
||||
#[serde(rename = "subjectNames")]
|
||||
pub subject_names: HashMap<String, String>,
|
||||
|
||||
#[serde(rename = "topicNames")]
|
||||
pub topic_names: HashMap<String, String>,
|
||||
|
||||
#[serde(rename = "topicSubTopics")]
|
||||
pub topic_subtopics: HashMap<String, Vec<i64>>,
|
||||
|
||||
#[serde(rename = "venueNames")]
|
||||
pub venue_names: HashMap<String, String>,
|
||||
|
||||
// Catch-all for other fields
|
||||
#[serde(flatten)]
|
||||
pub extra: HashMap<String, serde_json::Value>,
|
||||
pub events: HashMap<String, CITMEvent>,
|
||||
pub performances: Vec<CITMPerformance>,
|
||||
}
|
||||
|
||||
static mut CITM_DATA: *mut CitmCatalog = std::ptr::null_mut();
|
||||
|
||||
/// Creates a CitmCatalog from a JSON string (UTF-8 encoded).
|
||||
/// Only extracts events and performances to match C++ behavior.
|
||||
#[no_mangle]
|
||||
pub unsafe extern "C" fn citm_from_str(
|
||||
raw_input: *const c_char,
|
||||
@@ -414,7 +174,6 @@ pub unsafe extern "C" fn citm_from_str(
|
||||
return ptr::null_mut();
|
||||
}
|
||||
|
||||
// Convert the raw pointer + length into a Rust slice
|
||||
let bytes = slice::from_raw_parts(raw_input as *const u8, raw_input_length);
|
||||
let input_str = match std::str::from_utf8(bytes) {
|
||||
Ok(s) => s,
|
||||
@@ -424,12 +183,23 @@ pub unsafe extern "C" fn citm_from_str(
|
||||
}
|
||||
};
|
||||
|
||||
// Try deserializing the input string into CitmCatalog
|
||||
match serde_json::from_str::<CitmCatalog>(input_str) {
|
||||
Ok(catalog) => Box::into_raw(Box::new(catalog)),
|
||||
// Parse the full JSON to extract only events and performances
|
||||
match serde_json::from_str::<serde_json::Value>(input_str) {
|
||||
Ok(full_json) => {
|
||||
// Extract only the fields we need (matching C++ behavior)
|
||||
let events: HashMap<String, CITMEvent> = full_json.get("events")
|
||||
.and_then(|v| serde_json::from_value(v.clone()).ok())
|
||||
.unwrap_or_default();
|
||||
|
||||
let performances: Vec<CITMPerformance> = full_json.get("performances")
|
||||
.and_then(|v| serde_json::from_value(v.clone()).ok())
|
||||
.unwrap_or_default();
|
||||
|
||||
let catalog = CitmCatalog { events, performances };
|
||||
Box::into_raw(Box::new(catalog))
|
||||
},
|
||||
Err(e) => {
|
||||
eprintln!("Error deserializing JSON: {}", e);
|
||||
eprintln!("JSON snippet (first 200 chars): {:.200}...", input_str);
|
||||
ptr::null_mut()
|
||||
}
|
||||
}
|
||||
@@ -437,30 +207,17 @@ pub unsafe extern "C" fn citm_from_str(
|
||||
|
||||
/// Serializes a CitmCatalog into a JSON string (UTF-8).
|
||||
#[no_mangle]
|
||||
pub unsafe extern "C" fn str_from_citm(raw_catalog: *mut CitmCatalog) -> *mut c_char {
|
||||
if raw_catalog.is_null() {
|
||||
eprintln!("Error: Catalog pointer is null");
|
||||
return ptr::null_mut();
|
||||
}
|
||||
pub unsafe extern "C" fn set_citm_data(raw: *mut CitmCatalog) {
|
||||
CITM_DATA = raw;
|
||||
}
|
||||
|
||||
// Fix: Actually serialize the catalog
|
||||
let catalog = &*raw_catalog;
|
||||
|
||||
match serde_json::to_string(catalog) {
|
||||
Ok(serialized) => {
|
||||
match CString::new(serialized) {
|
||||
Ok(cstr) => cstr.into_raw(),
|
||||
Err(e) => {
|
||||
eprintln!("Error creating CString: {}", e);
|
||||
ptr::null_mut()
|
||||
}
|
||||
}
|
||||
},
|
||||
Err(e) => {
|
||||
eprintln!("Error serializing catalog to JSON: {}", e);
|
||||
ptr::null_mut()
|
||||
}
|
||||
#[no_mangle]
|
||||
pub unsafe extern "C" fn serialize_citm_to_string() -> usize {
|
||||
if CITM_DATA.is_null() {
|
||||
return 0;
|
||||
}
|
||||
let data = &*CITM_DATA;
|
||||
return serde_json::to_string(data).unwrap().len();
|
||||
}
|
||||
|
||||
/// Frees the CitmCatalog pointer.
|
||||
@@ -475,8 +232,120 @@ pub unsafe extern "C" fn free_citm(raw_catalog: *mut CitmCatalog) {
|
||||
pub extern "C" fn free_str(ptr: *mut c_char) {
|
||||
if !ptr.is_null() {
|
||||
unsafe {
|
||||
// Convert back into a CString, which automatically frees the memory
|
||||
let _ = CString::from_raw(ptr);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
//==============================================================================
|
||||
// FFI Overhead Measurement Functions
|
||||
// These allow measuring the actual FFI overhead vs pure Rust serialization
|
||||
//==============================================================================
|
||||
|
||||
/// Result structure for FFI overhead measurement
|
||||
#[repr(C)]
|
||||
pub struct FfiOverheadResult {
|
||||
/// Time in nanoseconds for pure serde_json::to_string() (no FFI overhead)
|
||||
pub pure_serde_ns: u64,
|
||||
/// Time in nanoseconds for serde + CString conversion
|
||||
pub serde_plus_cstring_ns: u64,
|
||||
/// Number of iterations performed
|
||||
pub iterations: u64,
|
||||
/// Output size in bytes (for verification)
|
||||
pub output_size: u64,
|
||||
}
|
||||
|
||||
/// Prevents compiler from optimizing away the value
|
||||
/// Works on stable Rust (unlike std::hint::black_box which is unstable)
|
||||
#[inline(never)]
|
||||
fn black_box<T>(dummy: T) -> T {
|
||||
unsafe {
|
||||
let ret = std::ptr::read_volatile(&dummy);
|
||||
std::mem::forget(dummy);
|
||||
ret
|
||||
}
|
||||
}
|
||||
|
||||
/// Measures FFI overhead for Twitter serialization.
|
||||
/// Performs `iterations` serializations entirely in Rust and returns timing data.
|
||||
/// This allows comparing against per-call FFI overhead.
|
||||
#[no_mangle]
|
||||
pub unsafe extern "C" fn measure_twitter_ffi_overhead(
|
||||
raw: *mut TwitterData,
|
||||
iterations: u64
|
||||
) -> FfiOverheadResult {
|
||||
use std::time::Instant;
|
||||
|
||||
let twitter_data = &*raw;
|
||||
let output_size: u64;
|
||||
|
||||
// Warm-up run
|
||||
let warmup = serde_json::to_string(&twitter_data).unwrap();
|
||||
output_size = warmup.len() as u64;
|
||||
|
||||
// Measure pure serde_json::to_string() - no CString conversion
|
||||
let start_pure = Instant::now();
|
||||
for _ in 0..iterations {
|
||||
let serialized = serde_json::to_string(&twitter_data).unwrap();
|
||||
// Prevent optimization from eliminating the work
|
||||
black_box(&serialized);
|
||||
}
|
||||
let pure_serde_ns = start_pure.elapsed().as_nanos() as u64;
|
||||
|
||||
// Measure serde + CString conversion (but not FFI return)
|
||||
let start_cstring = Instant::now();
|
||||
for _ in 0..iterations {
|
||||
let serialized = serde_json::to_string(&twitter_data).unwrap();
|
||||
let cstring = CString::new(serialized).unwrap();
|
||||
// Prevent optimization from eliminating the work
|
||||
black_box(&cstring);
|
||||
}
|
||||
let serde_plus_cstring_ns = start_cstring.elapsed().as_nanos() as u64;
|
||||
|
||||
FfiOverheadResult {
|
||||
pure_serde_ns,
|
||||
serde_plus_cstring_ns,
|
||||
iterations,
|
||||
output_size,
|
||||
}
|
||||
}
|
||||
|
||||
/// Measures FFI overhead for CITM serialization.
|
||||
#[no_mangle]
|
||||
pub unsafe extern "C" fn measure_citm_ffi_overhead(
|
||||
raw: *mut CitmCatalog,
|
||||
iterations: u64
|
||||
) -> FfiOverheadResult {
|
||||
use std::time::Instant;
|
||||
|
||||
let catalog = &*raw;
|
||||
let output_size: u64;
|
||||
|
||||
// Warm-up run
|
||||
let warmup = serde_json::to_string(&catalog).unwrap();
|
||||
output_size = warmup.len() as u64;
|
||||
|
||||
// Measure pure serde_json::to_string() - no CString conversion
|
||||
let start_pure = Instant::now();
|
||||
for _ in 0..iterations {
|
||||
let serialized = serde_json::to_string(&catalog).unwrap();
|
||||
black_box(&serialized);
|
||||
}
|
||||
let pure_serde_ns = start_pure.elapsed().as_nanos() as u64;
|
||||
|
||||
// Measure serde + CString conversion
|
||||
let start_cstring = Instant::now();
|
||||
for _ in 0..iterations {
|
||||
let serialized = serde_json::to_string(&catalog).unwrap();
|
||||
let cstring = CString::new(serialized).unwrap();
|
||||
black_box(&cstring);
|
||||
}
|
||||
let serde_plus_cstring_ns = start_cstring.elapsed().as_nanos() as u64;
|
||||
|
||||
FfiOverheadResult {
|
||||
pure_serde_ns,
|
||||
serde_plus_cstring_ns,
|
||||
iterations,
|
||||
output_size,
|
||||
}
|
||||
}
|
||||
|
||||
@@ -4,6 +4,7 @@
|
||||
/* Generated with cbindgen:0.28.0 */
|
||||
|
||||
/* Warning, this file is autogenerated by cbindgen. Don't modify this manually. */
|
||||
/* Note: FfiOverheadResult and measurement functions added manually */
|
||||
|
||||
#include <cstdarg>
|
||||
#include <cstdint>
|
||||
@@ -17,11 +18,25 @@ struct CitmCatalog;
|
||||
|
||||
struct TwitterData;
|
||||
|
||||
/// Result structure for FFI overhead measurement
|
||||
struct FfiOverheadResult {
|
||||
/// Time in nanoseconds for pure serde_json::to_string() (no FFI overhead)
|
||||
uint64_t pure_serde_ns;
|
||||
/// Time in nanoseconds for serde + CString conversion
|
||||
uint64_t serde_plus_cstring_ns;
|
||||
/// Number of iterations performed
|
||||
uint64_t iterations;
|
||||
/// Output size in bytes (for verification)
|
||||
uint64_t output_size;
|
||||
};
|
||||
|
||||
extern "C" {
|
||||
|
||||
TwitterData *twitter_from_str(const char *raw_input, size_t raw_input_length);
|
||||
|
||||
const char *str_from_twitter(TwitterData *raw);
|
||||
void set_twitter_data(TwitterData *raw);
|
||||
|
||||
size_t serialize_twitter_to_string();
|
||||
|
||||
void free_twitter(TwitterData *raw);
|
||||
|
||||
@@ -30,14 +45,22 @@ void free_string(const char *ptr);
|
||||
/// Creates a CitmCatalog from a JSON string (UTF-8 encoded).
|
||||
CitmCatalog *citm_from_str(const char *raw_input, uintptr_t raw_input_length);
|
||||
|
||||
/// Serializes a CitmCatalog into a JSON string (UTF-8).
|
||||
char *str_from_citm(CitmCatalog *raw_catalog);
|
||||
void set_citm_data(CitmCatalog *raw);
|
||||
|
||||
size_t serialize_citm_to_string();
|
||||
|
||||
/// Frees the CitmCatalog pointer.
|
||||
void free_citm(CitmCatalog *raw_catalog);
|
||||
|
||||
void free_str(char *ptr);
|
||||
|
||||
/// Measures FFI overhead for Twitter serialization.
|
||||
/// Performs `iterations` serializations entirely in Rust and returns timing data.
|
||||
FfiOverheadResult measure_twitter_ffi_overhead(TwitterData *raw, uint64_t iterations);
|
||||
|
||||
/// Measures FFI overhead for CITM serialization.
|
||||
FfiOverheadResult measure_citm_ffi_overhead(CitmCatalog *raw, uint64_t iterations);
|
||||
|
||||
} // extern "C"
|
||||
|
||||
} // namespace serde_benchmark
|
||||
|
||||
@@ -0,0 +1,31 @@
|
||||
use std::fs;
|
||||
|
||||
// Include the lib.rs content directly
|
||||
include!("../lib.rs");
|
||||
|
||||
fn main() {
|
||||
// Read the Twitter JSON file
|
||||
let json_str = fs::read_to_string("/Users/random_person/Desktop/simdjson/build/jsonexamples/twitter.json")
|
||||
.expect("Failed to read file");
|
||||
|
||||
// Parse it
|
||||
let data: TwitterData = serde_json::from_str(&json_str)
|
||||
.expect("Failed to parse JSON");
|
||||
|
||||
// Serialize it back
|
||||
let output = serde_json::to_string(&data)
|
||||
.expect("Failed to serialize");
|
||||
|
||||
// Write to file for comparison
|
||||
fs::write("rust_output.json", &output)
|
||||
.expect("Failed to write output");
|
||||
|
||||
println!("Output size: {} bytes", output.len());
|
||||
println!("Written to rust_output.json");
|
||||
|
||||
// Also write pretty version for easier inspection
|
||||
let pretty = serde_json::to_string_pretty(&data)
|
||||
.expect("Failed to serialize pretty");
|
||||
fs::write("rust_output_pretty.json", &pretty)
|
||||
.expect("Failed to write pretty output");
|
||||
}
|
||||
@@ -0,0 +1,36 @@
|
||||
use std::fs;
|
||||
|
||||
// Import from the parent lib.rs
|
||||
include!("../lib.rs");
|
||||
|
||||
fn main() {
|
||||
// Read the Twitter JSON file
|
||||
let json_str = fs::read_to_string("/Users/random_person/Desktop/simdjson/build/jsonexamples/twitter.json")
|
||||
.expect("Failed to read file");
|
||||
|
||||
// Parse it
|
||||
let data: TwitterData = serde_json::from_str(&json_str)
|
||||
.expect("Failed to parse JSON");
|
||||
|
||||
// Serialize it back (compact)
|
||||
let output = serde_json::to_vec(&data)
|
||||
.expect("Failed to serialize");
|
||||
|
||||
let output_str = String::from_utf8(output.clone()).unwrap();
|
||||
|
||||
// Write to file for comparison
|
||||
fs::write("rust_output_test.json", &output)
|
||||
.expect("Failed to write output");
|
||||
|
||||
println!("Output size: {} bytes", output.len());
|
||||
|
||||
// Count statuses
|
||||
println!("Number of statuses: {}", data.statuses.len());
|
||||
|
||||
// Check what fields are in the first status
|
||||
if let Some(first) = data.statuses.first() {
|
||||
// Let's serialize just the first status to see what fields are included
|
||||
let first_json = serde_json::to_string_pretty(first).unwrap();
|
||||
println!("First status (pretty):\n{}", first_json);
|
||||
}
|
||||
}
|
||||
@@ -1,14 +1,32 @@
|
||||
|
||||
# Add executable targets
|
||||
add_executable(benchmark_serialization_twitter benchmark_serialization_twitter.cpp)
|
||||
add_executable(benchmark_parsing_twitter benchmark_parsing_twitter.cpp)
|
||||
|
||||
if(TARGET serde-benchmark)
|
||||
message(STATUS "serde-benchmark target was created. Linking benchmarks and serde-benchmark.")
|
||||
target_link_libraries(benchmark_serialization_twitter PRIVATE serde-benchmark)
|
||||
target_link_libraries(benchmark_parsing_twitter PRIVATE serde-benchmark)
|
||||
target_compile_definitions(benchmark_serialization_twitter PRIVATE SIMDJSON_RUST_VERSION="${Rust_VERSION}")
|
||||
target_compile_definitions(benchmark_parsing_twitter PRIVATE SIMDJSON_RUST_VERSION="${Rust_VERSION}")
|
||||
endif()
|
||||
target_link_libraries(benchmark_serialization_twitter PRIVATE simdjson::simdjson nlohmann_json)
|
||||
target_link_libraries(benchmark_serialization_twitter PRIVATE reflectcpp)
|
||||
target_compile_definitions(benchmark_serialization_twitter PRIVATE SIMDJSON_BENCH_CPP_REFLECT=1)
|
||||
|
||||
target_compile_definitions(benchmark_serialization_twitter PRIVATE JSON_FILE="${EXAMPLE_JSON}")
|
||||
target_link_libraries(benchmark_parsing_twitter PRIVATE simdjson::simdjson nlohmann_json)
|
||||
|
||||
if(TARGET rapidjson)
|
||||
target_link_libraries(benchmark_parsing_twitter PRIVATE rapidjson)
|
||||
target_compile_definitions(benchmark_parsing_twitter PRIVATE SIMDJSON_COMPETITION_RAPIDJSON)
|
||||
endif()
|
||||
|
||||
if(TARGET yyjson)
|
||||
target_link_libraries(benchmark_parsing_twitter PRIVATE yyjson)
|
||||
target_compile_definitions(benchmark_parsing_twitter PRIVATE SIMDJSON_COMPETITION_YYJSON)
|
||||
target_link_libraries(benchmark_serialization_twitter PRIVATE yyjson)
|
||||
target_compile_definitions(benchmark_serialization_twitter PRIVATE SIMDJSON_COMPETITION_YYJSON)
|
||||
endif()
|
||||
|
||||
target_compile_definitions(benchmark_serialization_twitter PRIVATE JSON_FILE="${EXAMPLE_JSON}")
|
||||
target_compile_definitions(benchmark_parsing_twitter PRIVATE JSON_FILE="${EXAMPLE_JSON}")
|
||||
@@ -0,0 +1,232 @@
|
||||
#include <cassert>
|
||||
#include <cstdlib>
|
||||
#include <ctime>
|
||||
#include <format>
|
||||
#include <fstream>
|
||||
#include <iostream>
|
||||
#include <nlohmann/json.hpp>
|
||||
#include <simdjson.h>
|
||||
#include <string>
|
||||
#include "twitter_data.h"
|
||||
#include "nlohmann_twitter_data.h"
|
||||
#include "../benchmark_utils/benchmark_helper.h"
|
||||
#ifdef SIMDJSON_COMPETITION_RAPIDJSON
|
||||
#include "rapidjson_twitter_data.h"
|
||||
#endif
|
||||
#ifdef SIMDJSON_COMPETITION_YYJSON
|
||||
#include "yyjson_twitter_data.h"
|
||||
#endif
|
||||
|
||||
#ifdef SIMDJSON_RUST_VERSION
|
||||
#include "../serde-benchmark/serde_benchmark.h"
|
||||
|
||||
void bench_rust_parsing(const std::string &json_str) {
|
||||
size_t input_volume = json_str.size();
|
||||
printf("# input volume: %zu bytes\n", input_volume);
|
||||
|
||||
volatile bool result = true;
|
||||
pretty_print(1, input_volume, "bench_rust_parsing",
|
||||
bench([&json_str, &result]() {
|
||||
serde_benchmark::TwitterData *td = serde_benchmark::twitter_from_str(json_str.c_str(), json_str.size());
|
||||
result = (td != nullptr);
|
||||
if (td) {
|
||||
serde_benchmark::free_twitter(td);
|
||||
}
|
||||
if (!result) {
|
||||
printf("parse error\n");
|
||||
}
|
||||
}));
|
||||
}
|
||||
#endif
|
||||
|
||||
// OPTIMIZED VERSION: Reuses parser across iterations
|
||||
template <class T>
|
||||
void bench_simdjson_static_reflection_parsing(const std::string &json_str) {
|
||||
size_t input_volume = json_str.size();
|
||||
printf("# input volume: %zu bytes\n", input_volume);
|
||||
|
||||
// Pre-allocate padded buffer outside the benchmark loop
|
||||
simdjson::padded_string padded = simdjson::padded_string(json_str);
|
||||
|
||||
// CRITICAL: Create parser OUTSIDE the loop for reuse
|
||||
simdjson::ondemand::parser parser;
|
||||
|
||||
volatile bool result = true;
|
||||
pretty_print(1, input_volume, "bench_simdjson_static_reflection_parsing",
|
||||
bench([&padded, &result, &parser]() {
|
||||
// Reuse the same parser instance
|
||||
simdjson::ondemand::document doc;
|
||||
if(parser.iterate(padded).get(doc)) {
|
||||
result = false;
|
||||
return;
|
||||
}
|
||||
T my_struct;
|
||||
if(doc.get<T>().get(my_struct)) {
|
||||
result = false;
|
||||
}
|
||||
if (!result) {
|
||||
printf("parse error\n");
|
||||
}
|
||||
}));
|
||||
}
|
||||
|
||||
#if SIMDJSON_STATIC_REFLECTION
|
||||
template <class T>
|
||||
void bench_simdjson_from_parsing(const std::string &json_str) {
|
||||
size_t input_volume = json_str.size();
|
||||
printf("# input volume: %zu bytes\n", input_volume);
|
||||
|
||||
// Pre-allocate padded buffer outside the benchmark loop
|
||||
simdjson::padded_string padded = simdjson::padded_string(json_str);
|
||||
|
||||
volatile bool result = true;
|
||||
pretty_print(1, input_volume, "bench_simdjson_from_parsing",
|
||||
bench([&padded, &result]() {
|
||||
T my_struct;
|
||||
auto err = simdjson::from(padded).get(my_struct);
|
||||
if (err) {
|
||||
result = false;
|
||||
printf("parse error: %s\n", simdjson::error_message(err));
|
||||
return;
|
||||
}
|
||||
}));
|
||||
}
|
||||
#endif
|
||||
|
||||
void bench_nlohmann_parsing(const std::string &json_str) {
|
||||
size_t input_volume = json_str.size();
|
||||
printf("# input volume: %zu bytes\n", input_volume);
|
||||
|
||||
volatile bool result = true;
|
||||
pretty_print(1, input_volume, "bench_nlohmann_parsing",
|
||||
bench([&json_str, &result]() {
|
||||
try {
|
||||
TwitterData data = nlohmann_deserialize(json_str);
|
||||
result = true;
|
||||
} catch (...) {
|
||||
result = false;
|
||||
printf("parse error\n");
|
||||
}
|
||||
}));
|
||||
}
|
||||
|
||||
#ifdef SIMDJSON_COMPETITION_RAPIDJSON
|
||||
void bench_rapidjson_parsing(const std::string &json_str) {
|
||||
size_t input_volume = json_str.size();
|
||||
printf("# input volume: %zu bytes\n", input_volume);
|
||||
|
||||
volatile bool result = true;
|
||||
pretty_print(1, input_volume, "bench_rapidjson_parsing",
|
||||
bench([&json_str, &result]() {
|
||||
try {
|
||||
TwitterData data = rapidjson_deserialize(json_str);
|
||||
result = true;
|
||||
} catch (...) {
|
||||
result = false;
|
||||
printf("parse error\n");
|
||||
}
|
||||
}));
|
||||
}
|
||||
#endif
|
||||
|
||||
#ifdef SIMDJSON_COMPETITION_YYJSON
|
||||
void bench_yyjson_parsing(const std::string &json_str) {
|
||||
size_t input_volume = json_str.size();
|
||||
printf("# input volume: %zu bytes\n", input_volume);
|
||||
|
||||
volatile bool result = true;
|
||||
pretty_print(1, input_volume, "bench_yyjson_parsing",
|
||||
bench([&json_str, &result]() {
|
||||
try {
|
||||
TwitterData data = yyjson_deserialize(json_str);
|
||||
result = true;
|
||||
} catch (...) {
|
||||
result = false;
|
||||
printf("parse error\n");
|
||||
}
|
||||
}));
|
||||
}
|
||||
#endif
|
||||
|
||||
std::string read_file(std::string filename) {
|
||||
printf("# Reading file %s\n", filename.c_str());
|
||||
constexpr size_t read_size = 4096;
|
||||
auto stream = std::ifstream(filename.c_str());
|
||||
stream.exceptions(std::ios_base::badbit);
|
||||
std::string out;
|
||||
std::string buf(read_size, '\0');
|
||||
while (stream.read(&buf[0], read_size)) {
|
||||
out.append(buf, 0, size_t(stream.gcount()));
|
||||
}
|
||||
out.append(buf, 0, size_t(stream.gcount()));
|
||||
return out;
|
||||
}
|
||||
|
||||
// Function to check if benchmark name matches any of the comma-separated filters
|
||||
bool matches_filter(const std::string& benchmark_name, const std::string& filter) {
|
||||
if (filter.empty()) return true;
|
||||
|
||||
// Split filter by comma
|
||||
size_t start = 0;
|
||||
size_t end = filter.find(',');
|
||||
while (end != std::string::npos) {
|
||||
std::string token = filter.substr(start, end - start);
|
||||
if (benchmark_name.find(token) != std::string::npos) {
|
||||
return true;
|
||||
}
|
||||
start = end + 1;
|
||||
end = filter.find(',', start);
|
||||
}
|
||||
// Check last token
|
||||
std::string token = filter.substr(start);
|
||||
return benchmark_name.find(token) != std::string::npos;
|
||||
}
|
||||
|
||||
int main(int argc, char* argv[]) {
|
||||
std::string filter;
|
||||
|
||||
// Parse command-line arguments
|
||||
for (int i = 1; i < argc; ++i) {
|
||||
if (strcmp(argv[i], "-f") == 0 || strcmp(argv[i], "--filter") == 0) {
|
||||
if (i + 1 < argc) {
|
||||
filter = argv[++i];
|
||||
} else {
|
||||
std::cerr << "Error: -f/--filter requires an argument" << std::endl;
|
||||
return EXIT_FAILURE;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Load the JSON data
|
||||
std::string json_str = read_file(JSON_FILE);
|
||||
|
||||
// Benchmarking the parsing
|
||||
if (matches_filter("nlohmann", filter)) {
|
||||
bench_nlohmann_parsing(json_str);
|
||||
}
|
||||
#ifdef SIMDJSON_COMPETITION_RAPIDJSON
|
||||
if (matches_filter("rapidjson", filter)) {
|
||||
bench_rapidjson_parsing(json_str);
|
||||
}
|
||||
#endif
|
||||
#ifdef SIMDJSON_COMPETITION_YYJSON
|
||||
if (matches_filter("yyjson", filter)) {
|
||||
bench_yyjson_parsing(json_str);
|
||||
}
|
||||
#endif
|
||||
if (matches_filter("simdjson_static_reflection", filter)) {
|
||||
bench_simdjson_static_reflection_parsing<TwitterData>(json_str);
|
||||
}
|
||||
#if SIMDJSON_STATIC_REFLECTION
|
||||
if (matches_filter("simdjson_from", filter)) {
|
||||
bench_simdjson_from_parsing<TwitterData>(json_str);
|
||||
}
|
||||
#endif
|
||||
#ifdef SIMDJSON_RUST_VERSION
|
||||
if (matches_filter("rust", filter)) {
|
||||
printf("# Note: Rust/Serde parsing test\n");
|
||||
bench_rust_parsing(json_str);
|
||||
}
|
||||
#endif
|
||||
return EXIT_SUCCESS;
|
||||
}
|
||||
@@ -1,4 +1,5 @@
|
||||
#include <cassert>
|
||||
#include <chrono>
|
||||
#include <cstdlib>
|
||||
#include <ctime>
|
||||
#include <format>
|
||||
@@ -10,6 +11,9 @@
|
||||
#include "twitter_data.h"
|
||||
#include "nlohmann_twitter_data.h"
|
||||
#include "../benchmark_utils/benchmark_helper.h"
|
||||
#ifdef SIMDJSON_COMPETITION_YYJSON
|
||||
#include "yyjson_twitter_data.h"
|
||||
#endif
|
||||
#if SIMDJSON_BENCH_CPP_REFLECT
|
||||
#include <rfl.hpp>
|
||||
#include <rfl/json.hpp>
|
||||
@@ -35,19 +39,49 @@ void bench_reflect_cpp(TwitterData &data) {
|
||||
|
||||
|
||||
void bench_rust(serde_benchmark::TwitterData *data) {
|
||||
const char * output = serde_benchmark::str_from_twitter(data);
|
||||
size_t output_volume = strlen(output);
|
||||
serde_benchmark::set_twitter_data(data);
|
||||
size_t output_volume = serde_benchmark::serialize_twitter_to_string();
|
||||
printf("# output volume: %zu bytes\n", output_volume);
|
||||
|
||||
volatile size_t measured_volume = 0;
|
||||
pretty_print(1, output_volume, "bench_rust",
|
||||
bench([&data, &measured_volume, &output_volume]() {
|
||||
const char * output = serde_benchmark::str_from_twitter(data);
|
||||
serde_benchmark::free_string(output);
|
||||
bench([&measured_volume, &output_volume]() {
|
||||
measured_volume = serde_benchmark::serialize_twitter_to_string();
|
||||
}));
|
||||
}
|
||||
#endif
|
||||
|
||||
// Fair allocation variant: allocates fresh buffer each iteration (matches other libraries)
|
||||
template <class T> void bench_simdjson_static_reflection(T &data) {
|
||||
// First run to determine expected size
|
||||
simdjson::builder::string_builder sb_init;
|
||||
simdjson::builder::append(sb_init, data);
|
||||
std::string_view p_init;
|
||||
if(sb_init.view().get(p_init)) {
|
||||
std::cerr << "Error!" << std::endl;
|
||||
}
|
||||
size_t output_volume = p_init.size();
|
||||
printf("# output volume: %zu bytes\n", output_volume);
|
||||
|
||||
volatile size_t measured_volume = 0;
|
||||
pretty_print(sizeof(data), output_volume, "bench_simdjson_static_reflection",
|
||||
bench([&data, &measured_volume, &output_volume]() {
|
||||
// Fresh allocation each iteration - fair comparison
|
||||
simdjson::builder::string_builder sb;
|
||||
simdjson::builder::append(sb, data);
|
||||
std::string_view p;
|
||||
if(sb.view().get(p)) {
|
||||
std::cerr << "Error!" << std::endl;
|
||||
}
|
||||
measured_volume = sb.size();
|
||||
if (measured_volume != output_volume) {
|
||||
printf("mismatch\n");
|
||||
}
|
||||
}));
|
||||
}
|
||||
|
||||
// Optimized variant: reuses buffer across iterations (shows API potential)
|
||||
template <class T> void bench_simdjson_static_reflection_reuse(T &data) {
|
||||
simdjson::builder::string_builder sb;
|
||||
simdjson::builder::append(sb, data);
|
||||
std::string_view p;
|
||||
@@ -59,7 +93,7 @@ template <class T> void bench_simdjson_static_reflection(T &data) {
|
||||
printf("# output volume: %zu bytes\n", output_volume);
|
||||
|
||||
volatile size_t measured_volume = 0;
|
||||
pretty_print(sizeof(data), output_volume, "bench_simdjson_static_reflection",
|
||||
pretty_print(sizeof(data), output_volume, "bench_simdjson_reuse_buffer",
|
||||
bench([&data, &measured_volume, &output_volume, &sb]() {
|
||||
sb.clear();
|
||||
simdjson::builder::append(sb, data);
|
||||
@@ -74,6 +108,63 @@ template <class T> void bench_simdjson_static_reflection(T &data) {
|
||||
}));
|
||||
}
|
||||
|
||||
#if SIMDJSON_STATIC_REFLECTION
|
||||
// Fair allocation variant: allocates fresh string each iteration
|
||||
template <class T> void bench_simdjson_to(T &data) {
|
||||
// First run to determine size
|
||||
std::string output_init;
|
||||
if (simdjson::error_code err = simdjson::builder::to_json(data, output_init); err) {
|
||||
std::cerr << "Error in to_json initialization!" << simdjson::error_message(err) << std::endl;
|
||||
return;
|
||||
}
|
||||
size_t output_volume = output_init.size();
|
||||
printf("# output volume: %zu bytes\n", output_volume);
|
||||
|
||||
volatile size_t measured_volume = 0;
|
||||
pretty_print(sizeof(data), output_volume, "bench_simdjson_to",
|
||||
bench([&data, &measured_volume, &output_volume]() {
|
||||
// Fresh allocation each iteration - fair comparison
|
||||
std::string output;
|
||||
if (simdjson::error_code err = simdjson::builder::to_json(data, output); err) {
|
||||
std::cerr << "Error in to_json!" << simdjson::error_message(err) << std::endl;
|
||||
return;
|
||||
}
|
||||
measured_volume = output.size();
|
||||
if (measured_volume != output_volume) {
|
||||
printf("mismatch\n");
|
||||
}
|
||||
}));
|
||||
}
|
||||
|
||||
// Optimized variant: reuses pre-allocated string
|
||||
template <class T> void bench_simdjson_to_reuse(T &data) {
|
||||
std::string output;
|
||||
if (simdjson::error_code err = simdjson::builder::to_json(data, output); err) {
|
||||
std::cerr << "Error in to_json initialization!" << simdjson::error_message(err) << std::endl;
|
||||
return;
|
||||
}
|
||||
size_t output_volume = output.size();
|
||||
printf("# output volume: %zu bytes\n", output_volume);
|
||||
|
||||
// Pre-allocate string with sufficient capacity to avoid reallocation
|
||||
output.reserve(output_volume * 2);
|
||||
|
||||
volatile size_t measured_volume = 0;
|
||||
pretty_print(sizeof(data), output_volume, "bench_simdjson_to_reuse",
|
||||
bench([&data, &measured_volume, &output_volume, &output]() {
|
||||
// Reuse the pre-allocated string - avoids allocation
|
||||
if (simdjson::error_code err = simdjson::builder::to_json(data, output); err) {
|
||||
std::cerr << "Error in to_json!" << simdjson::error_message(err) << std::endl;
|
||||
return;
|
||||
}
|
||||
measured_volume = output.size();
|
||||
if (measured_volume != output_volume) {
|
||||
printf("mismatch\n");
|
||||
}
|
||||
}));
|
||||
}
|
||||
#endif
|
||||
|
||||
void bench_nlohmann(TwitterData &data) {
|
||||
std::string output = nlohmann_serialize(data);
|
||||
size_t output_volume = output.size();
|
||||
@@ -90,28 +181,61 @@ void bench_nlohmann(TwitterData &data) {
|
||||
}));
|
||||
}
|
||||
|
||||
#ifdef SIMDJSON_COMPETITION_YYJSON
|
||||
void bench_yyjson(TwitterData &data) {
|
||||
std::string output = yyjson_serialize(data);
|
||||
size_t output_volume = output.size();
|
||||
printf("# output volume: %zu bytes\n", output_volume);
|
||||
|
||||
volatile size_t measured_volume = 0;
|
||||
pretty_print(1, output_volume, "bench_yyjson",
|
||||
bench([&data, &measured_volume, &output_volume]() {
|
||||
std::string output = yyjson_serialize(data);
|
||||
measured_volume = output.size();
|
||||
if (measured_volume != output_volume) {
|
||||
printf("mismatch\n");
|
||||
}
|
||||
}));
|
||||
}
|
||||
#endif
|
||||
|
||||
size_t WriteCallback(void *contents, size_t size, size_t nmemb, void *userp) {
|
||||
((std::string *)userp)->append((char *)contents, size * nmemb);
|
||||
return size * nmemb;
|
||||
}
|
||||
|
||||
std::string read_file(std::string filename) {
|
||||
simdjson::padded_string read_file(std::string filename) {
|
||||
printf("# Reading file %s\n", filename.c_str());
|
||||
constexpr size_t read_size = 4096;
|
||||
auto stream = std::ifstream(filename.c_str());
|
||||
stream.exceptions(std::ios_base::badbit);
|
||||
std::string out;
|
||||
simdjson::padded_string_builder builder;
|
||||
std::string buf(read_size, '\0');
|
||||
while (stream.read(&buf[0], read_size)) {
|
||||
out.append(buf, 0, size_t(stream.gcount()));
|
||||
builder.append(buf.data(), size_t(stream.gcount()));
|
||||
}
|
||||
out.append(buf, 0, size_t(stream.gcount()));
|
||||
return out;
|
||||
builder.append(buf.data(), size_t(stream.gcount()));
|
||||
return builder.convert();
|
||||
}
|
||||
|
||||
// Function to check if benchmark name contains filter substring
|
||||
// Function to check if benchmark name matches any of the comma-separated filters
|
||||
bool matches_filter(const std::string& benchmark_name, const std::string& filter) {
|
||||
return filter.empty() || benchmark_name.find(filter) != std::string::npos;
|
||||
if (filter.empty()) return true;
|
||||
|
||||
// Split filter by comma
|
||||
size_t start = 0;
|
||||
size_t end = filter.find(',');
|
||||
while (end != std::string::npos) {
|
||||
std::string token = filter.substr(start, end - start);
|
||||
if (benchmark_name.find(token) != std::string::npos) {
|
||||
return true;
|
||||
}
|
||||
start = end + 1;
|
||||
end = filter.find(',', start);
|
||||
}
|
||||
// Check last token
|
||||
std::string token = filter.substr(start);
|
||||
return benchmark_name.find(token) != std::string::npos;
|
||||
}
|
||||
|
||||
int main(int argc, char* argv[]) {
|
||||
@@ -129,12 +253,12 @@ int main(int argc, char* argv[]) {
|
||||
}
|
||||
}
|
||||
// Testing correctness of round-trip (serialization + deserialization)
|
||||
std::string json_str = read_file(JSON_FILE);
|
||||
simdjson::padded_string json_str = read_file(JSON_FILE);
|
||||
|
||||
// Loading up the data into a structure.
|
||||
simdjson::ondemand::parser parser;
|
||||
simdjson::ondemand::document doc;
|
||||
if(parser.iterate(simdjson::pad(json_str)).get(doc)) {
|
||||
if(parser.iterate(json_str).get(doc)) {
|
||||
std::cerr << "Error loading the document!" << std::endl;
|
||||
return EXIT_FAILURE;
|
||||
}
|
||||
@@ -145,18 +269,41 @@ int main(int argc, char* argv[]) {
|
||||
}
|
||||
|
||||
// Benchmarking the serialization
|
||||
// Note: simdjson benchmarks include both "fair" (fresh allocation) and "reuse" (buffer reuse) variants
|
||||
// The "fair" variants allocate fresh memory each iteration, matching other libraries' behavior
|
||||
// The "reuse" variants demonstrate the API's potential when buffer reuse is possible
|
||||
|
||||
if (matches_filter("nlohmann", filter)) {
|
||||
bench_nlohmann(my_struct);
|
||||
}
|
||||
#ifdef SIMDJSON_COMPETITION_YYJSON
|
||||
if (matches_filter("yyjson", filter)) {
|
||||
bench_yyjson(my_struct);
|
||||
}
|
||||
#endif
|
||||
if (matches_filter("simdjson_static_reflection", filter)) {
|
||||
bench_simdjson_static_reflection(my_struct);
|
||||
}
|
||||
if (matches_filter("simdjson_reuse", filter)) {
|
||||
bench_simdjson_static_reflection_reuse(my_struct);
|
||||
}
|
||||
#if SIMDJSON_STATIC_REFLECTION
|
||||
if (matches_filter("simdjson_to", filter)) {
|
||||
bench_simdjson_to(my_struct);
|
||||
}
|
||||
if (matches_filter("simdjson_to_reuse", filter)) {
|
||||
bench_simdjson_to_reuse(my_struct);
|
||||
}
|
||||
#endif
|
||||
#ifdef SIMDJSON_RUST_VERSION
|
||||
if (matches_filter("rust", filter)) {
|
||||
printf("# WARNING: The Rust benchmark may not be directly comparable since it does not use an equivalent data structure.\n");
|
||||
serde_benchmark::TwitterData * td = serde_benchmark::twitter_from_str(json_str.c_str(), json_str.size());
|
||||
bench_rust(td);
|
||||
serde_benchmark::free_twitter(td);
|
||||
serde_benchmark::TwitterData * td = serde_benchmark::twitter_from_str(json_str.data(), json_str.size());
|
||||
if (td == nullptr) {
|
||||
printf("# Failed to parse Twitter data for Rust benchmark\n");
|
||||
} else {
|
||||
bench_rust(td);
|
||||
serde_benchmark::free_twitter(td);
|
||||
}
|
||||
}
|
||||
#endif
|
||||
#if SIMDJSON_BENCH_CPP_REFLECT
|
||||
|
||||
@@ -4,6 +4,7 @@
|
||||
#include "twitter_data.h"
|
||||
#include <nlohmann/json.hpp>
|
||||
|
||||
// Serialization functions for nlohmann
|
||||
void to_json(nlohmann::json &j, const User &u) {
|
||||
j = nlohmann::json{{"id", u.id},
|
||||
{"name", u.name},
|
||||
@@ -16,101 +17,54 @@ void to_json(nlohmann::json &j, const User &u) {
|
||||
{"statuses_count", u.statuses_count}};
|
||||
}
|
||||
|
||||
void to_json(nlohmann::json &j, const Hashtag &h) {
|
||||
j = nlohmann::json{{"text", h.text},
|
||||
{"indices_start", h.indices_start},
|
||||
{"indices_end", h.indices_end}};
|
||||
}
|
||||
|
||||
void to_json(nlohmann::json &j, const Url &u) {
|
||||
j = nlohmann::json{{"url", u.url},
|
||||
{"expanded_url", u.expanded_url},
|
||||
{"display_url", u.display_url},
|
||||
{"indices_start", u.indices_start},
|
||||
{"indices_end", u.indices_end}};
|
||||
}
|
||||
|
||||
void to_json(nlohmann::json &j, const UserMention &um) {
|
||||
j = nlohmann::json{{"id", um.id},
|
||||
{"name", um.name},
|
||||
{"screen_name", um.screen_name},
|
||||
{"indices_start", um.indices_start},
|
||||
{"indices_end", um.indices_end}};
|
||||
}
|
||||
|
||||
void to_json(nlohmann::json &j, const Entities &e) {
|
||||
j = nlohmann::json{{"hashtags", e.hashtags},
|
||||
{"urls", e.urls},
|
||||
{"user_mentions", e.user_mentions}};
|
||||
}
|
||||
|
||||
void to_json(nlohmann::json &j, const Status &s) {
|
||||
j = nlohmann::json{{"created_at", s.created_at},
|
||||
{"id", s.id},
|
||||
{"text", s.text},
|
||||
{"user", s.user},
|
||||
{"entities", s.entities},
|
||||
{"retweet_count", s.retweet_count},
|
||||
{"favorite_count", s.favorite_count},
|
||||
{"favorited", s.favorited},
|
||||
{"retweeted", s.retweeted}};
|
||||
}
|
||||
|
||||
|
||||
std::string nlohmann_serialize(const std::vector<Hashtag>& v) {
|
||||
nlohmann::json a = nlohmann::json::array();
|
||||
for(const Hashtag & h : v) {
|
||||
a.push_back(nlohmann::json{{"text", h.text},
|
||||
{"indices_start", h.indices_start},
|
||||
{"indices_end", h.indices_end}});
|
||||
}
|
||||
return a.dump();
|
||||
}
|
||||
std::string nlohmann_serialize(const std::vector<Url>& v) {
|
||||
nlohmann::json a = nlohmann::json::array();
|
||||
for(const Url & u : v) {
|
||||
a.push_back(nlohmann::json{{"url", u.url},
|
||||
{"expanded_url", u.expanded_url},
|
||||
{"display_url", u.display_url},
|
||||
{"indices_start", u.indices_start},
|
||||
{"indices_end", u.indices_end}});
|
||||
}
|
||||
return a.dump();
|
||||
}
|
||||
std::string nlohmann_serialize(const std::vector<UserMention>& v) {
|
||||
nlohmann::json a = nlohmann::json::array();
|
||||
for(const UserMention & um : v) {
|
||||
a.push_back(nlohmann::json{{"id", um.id},
|
||||
{"name", um.name},
|
||||
{"screen_name", um.screen_name},
|
||||
{"indices_start", um.indices_start},
|
||||
{"indices_end", um.indices_end}});
|
||||
}
|
||||
return a.dump();
|
||||
}
|
||||
|
||||
std::string nlohmann_serialize(const std::vector<Status>& v) {
|
||||
nlohmann::json a = nlohmann::json::array();
|
||||
for(const Status & s : v) {
|
||||
a.push_back(nlohmann::json{{"created_at", s.created_at},
|
||||
{"id", s.id},
|
||||
{"text", s.text},
|
||||
{"user", s.user},
|
||||
{"entities", s.entities},
|
||||
{"retweet_count", s.retweet_count},
|
||||
{"favorite_count", s.favorite_count},
|
||||
{"favorited", s.favorited},
|
||||
{"retweeted", s.retweeted}});
|
||||
}
|
||||
return a.dump();
|
||||
{"favorite_count", s.favorite_count}};
|
||||
}
|
||||
|
||||
void to_json(nlohmann::json &j, const TwitterData &t) {
|
||||
j = nlohmann::json{{"statuses", t.statuses}};
|
||||
}
|
||||
|
||||
// Deserialization functions for nlohmann
|
||||
void from_json(const nlohmann::json &j, User &u) {
|
||||
j.at("id").get_to(u.id);
|
||||
j.at("name").get_to(u.name);
|
||||
j.at("screen_name").get_to(u.screen_name);
|
||||
j.at("location").get_to(u.location);
|
||||
j.at("description").get_to(u.description);
|
||||
j.at("verified").get_to(u.verified);
|
||||
j.at("followers_count").get_to(u.followers_count);
|
||||
j.at("friends_count").get_to(u.friends_count);
|
||||
j.at("statuses_count").get_to(u.statuses_count);
|
||||
}
|
||||
|
||||
void from_json(const nlohmann::json &j, Status &s) {
|
||||
j.at("created_at").get_to(s.created_at);
|
||||
j.at("id").get_to(s.id);
|
||||
j.at("text").get_to(s.text);
|
||||
j.at("user").get_to(s.user);
|
||||
j.at("retweet_count").get_to(s.retweet_count);
|
||||
j.at("favorite_count").get_to(s.favorite_count);
|
||||
}
|
||||
|
||||
void from_json(const nlohmann::json &j, TwitterData &t) {
|
||||
j.at("statuses").get_to(t.statuses);
|
||||
}
|
||||
|
||||
// Helper functions for benchmarking
|
||||
std::string nlohmann_serialize(const TwitterData &data) {
|
||||
return nlohmann_serialize(data.statuses);
|
||||
nlohmann::json j = data;
|
||||
return j.dump();
|
||||
}
|
||||
|
||||
TwitterData nlohmann_deserialize(const std::string &json_str) {
|
||||
nlohmann::json j = nlohmann::json::parse(json_str);
|
||||
return j.get<TwitterData>();
|
||||
}
|
||||
|
||||
#endif // NLOHMANN_TWITTER_DATA_H
|
||||
@@ -0,0 +1,142 @@
|
||||
#ifndef RAPIDJSON_TWITTER_DATA_H
|
||||
#define RAPIDJSON_TWITTER_DATA_H
|
||||
|
||||
#include "twitter_data.h"
|
||||
#include <rapidjson/document.h>
|
||||
#include <rapidjson/writer.h>
|
||||
#include <rapidjson/stringbuffer.h>
|
||||
#include <rapidjson/error/en.h>
|
||||
|
||||
using namespace rapidjson;
|
||||
|
||||
// RapidJSON deserialization for simplified Twitter data
|
||||
TwitterData rapidjson_deserialize(const std::string& json_str) {
|
||||
Document doc;
|
||||
doc.Parse(json_str.c_str());
|
||||
|
||||
if (doc.HasParseError()) {
|
||||
throw std::runtime_error("RapidJSON parse error");
|
||||
}
|
||||
|
||||
TwitterData data;
|
||||
|
||||
if (!doc.HasMember("statuses") || !doc["statuses"].IsArray()) {
|
||||
return data;
|
||||
}
|
||||
|
||||
const Value& statuses = doc["statuses"];
|
||||
data.statuses.reserve(statuses.Size());
|
||||
|
||||
for (SizeType i = 0; i < statuses.Size(); i++) {
|
||||
const Value& status_json = statuses[i];
|
||||
Status status;
|
||||
|
||||
// Parse status fields
|
||||
if (status_json.HasMember("created_at") && status_json["created_at"].IsString())
|
||||
status.created_at = status_json["created_at"].GetString();
|
||||
if (status_json.HasMember("id") && status_json["id"].IsUint64())
|
||||
status.id = status_json["id"].GetUint64();
|
||||
if (status_json.HasMember("text") && status_json["text"].IsString())
|
||||
status.text = status_json["text"].GetString();
|
||||
if (status_json.HasMember("retweet_count") && status_json["retweet_count"].IsUint64())
|
||||
status.retweet_count = status_json["retweet_count"].GetUint64();
|
||||
if (status_json.HasMember("favorite_count") && status_json["favorite_count"].IsUint64())
|
||||
status.favorite_count = status_json["favorite_count"].GetUint64();
|
||||
|
||||
// Parse user
|
||||
if (status_json.HasMember("user") && status_json["user"].IsObject()) {
|
||||
const Value& user_json = status_json["user"];
|
||||
User user;
|
||||
|
||||
if (user_json.HasMember("id") && user_json["id"].IsUint64())
|
||||
user.id = user_json["id"].GetUint64();
|
||||
if (user_json.HasMember("name") && user_json["name"].IsString())
|
||||
user.name = user_json["name"].GetString();
|
||||
if (user_json.HasMember("screen_name") && user_json["screen_name"].IsString())
|
||||
user.screen_name = user_json["screen_name"].GetString();
|
||||
if (user_json.HasMember("location") && user_json["location"].IsString())
|
||||
user.location = user_json["location"].GetString();
|
||||
if (user_json.HasMember("description") && user_json["description"].IsString())
|
||||
user.description = user_json["description"].GetString();
|
||||
if (user_json.HasMember("verified") && user_json["verified"].IsBool())
|
||||
user.verified = user_json["verified"].GetBool();
|
||||
if (user_json.HasMember("followers_count") && user_json["followers_count"].IsUint64())
|
||||
user.followers_count = user_json["followers_count"].GetUint64();
|
||||
if (user_json.HasMember("friends_count") && user_json["friends_count"].IsUint64())
|
||||
user.friends_count = user_json["friends_count"].GetUint64();
|
||||
if (user_json.HasMember("statuses_count") && user_json["statuses_count"].IsUint64())
|
||||
user.statuses_count = user_json["statuses_count"].GetUint64();
|
||||
|
||||
status.user = user;
|
||||
}
|
||||
|
||||
data.statuses.push_back(status);
|
||||
}
|
||||
|
||||
return data;
|
||||
}
|
||||
|
||||
// RapidJSON serialization for simplified Twitter data
|
||||
std::string rapidjson_serialize(const TwitterData& data) {
|
||||
Document doc;
|
||||
doc.SetObject();
|
||||
Document::AllocatorType& allocator = doc.GetAllocator();
|
||||
|
||||
Value statuses_array(kArrayType);
|
||||
|
||||
for (const auto& status : data.statuses) {
|
||||
Value status_obj(kObjectType);
|
||||
|
||||
Value created_at;
|
||||
created_at.SetString(status.created_at.c_str(), allocator);
|
||||
status_obj.AddMember("created_at", created_at, allocator);
|
||||
|
||||
status_obj.AddMember("id", status.id, allocator);
|
||||
|
||||
Value text;
|
||||
text.SetString(status.text.c_str(), allocator);
|
||||
status_obj.AddMember("text", text, allocator);
|
||||
|
||||
// Add user
|
||||
Value user_obj(kObjectType);
|
||||
user_obj.AddMember("id", status.user.id, allocator);
|
||||
|
||||
Value name;
|
||||
name.SetString(status.user.name.c_str(), allocator);
|
||||
user_obj.AddMember("name", name, allocator);
|
||||
|
||||
Value screen_name;
|
||||
screen_name.SetString(status.user.screen_name.c_str(), allocator);
|
||||
user_obj.AddMember("screen_name", screen_name, allocator);
|
||||
|
||||
Value location;
|
||||
location.SetString(status.user.location.c_str(), allocator);
|
||||
user_obj.AddMember("location", location, allocator);
|
||||
|
||||
Value description;
|
||||
description.SetString(status.user.description.c_str(), allocator);
|
||||
user_obj.AddMember("description", description, allocator);
|
||||
|
||||
user_obj.AddMember("verified", status.user.verified, allocator);
|
||||
user_obj.AddMember("followers_count", status.user.followers_count, allocator);
|
||||
user_obj.AddMember("friends_count", status.user.friends_count, allocator);
|
||||
user_obj.AddMember("statuses_count", status.user.statuses_count, allocator);
|
||||
|
||||
status_obj.AddMember("user", user_obj, allocator);
|
||||
|
||||
status_obj.AddMember("retweet_count", status.retweet_count, allocator);
|
||||
status_obj.AddMember("favorite_count", status.favorite_count, allocator);
|
||||
|
||||
statuses_array.PushBack(status_obj, allocator);
|
||||
}
|
||||
|
||||
doc.AddMember("statuses", statuses_array, allocator);
|
||||
|
||||
StringBuffer buffer;
|
||||
Writer<StringBuffer> writer(buffer);
|
||||
doc.Accept(writer);
|
||||
|
||||
return buffer.GetString();
|
||||
}
|
||||
|
||||
#endif // RAPIDJSON_TWITTER_DATA_H
|
||||
@@ -4,68 +4,31 @@
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
// Simplified Twitter structures for benchmarking
|
||||
|
||||
struct User {
|
||||
int64_t id;
|
||||
std::string id_str;
|
||||
uint64_t id;
|
||||
std::string name;
|
||||
std::string screen_name;
|
||||
std::string location;
|
||||
std::string description;
|
||||
bool verified;
|
||||
int64_t followers_count;
|
||||
int64_t friends_count;
|
||||
int64_t statuses_count;
|
||||
bool operator<=>(const User &other) const = default;
|
||||
};
|
||||
|
||||
struct Hashtag {
|
||||
std::string text;
|
||||
int64_t indices_start;
|
||||
int64_t indices_end;
|
||||
bool operator<=>(const Hashtag &other) const = default;
|
||||
};
|
||||
|
||||
struct Url {
|
||||
std::string url;
|
||||
std::string expanded_url;
|
||||
std::string display_url;
|
||||
int64_t indices_start;
|
||||
int64_t indices_end;
|
||||
bool operator<=>(const Url &other) const = default;
|
||||
};
|
||||
|
||||
struct UserMention {
|
||||
int64_t id;
|
||||
std::string name;
|
||||
std::string screen_name;
|
||||
int64_t indices_start;
|
||||
int64_t indices_end;
|
||||
bool operator<=>(const UserMention &other) const = default;
|
||||
};
|
||||
|
||||
struct Entities {
|
||||
std::vector<Hashtag> hashtags;
|
||||
std::vector<Url> urls;
|
||||
std::vector<UserMention> user_mentions;
|
||||
bool operator==(const Entities &other) const = default;
|
||||
uint64_t followers_count;
|
||||
uint64_t friends_count;
|
||||
uint64_t statuses_count;
|
||||
};
|
||||
|
||||
struct Status {
|
||||
std::string created_at;
|
||||
int64_t id;
|
||||
uint64_t id;
|
||||
std::string text;
|
||||
User user;
|
||||
Entities entities;
|
||||
int64_t retweet_count;
|
||||
int64_t favorite_count;
|
||||
bool favorited;
|
||||
bool retweeted;
|
||||
bool operator==(const Status &other) const = default;
|
||||
uint64_t retweet_count;
|
||||
uint64_t favorite_count;
|
||||
};
|
||||
|
||||
struct TwitterData {
|
||||
std::vector<Status> statuses;
|
||||
bool operator==(const TwitterData &other) const = default;
|
||||
};
|
||||
|
||||
#endif
|
||||
#endif // TWITTER_DATA_H
|
||||
@@ -0,0 +1,145 @@
|
||||
#ifndef YYJSON_TWITTER_DATA_H
|
||||
#define YYJSON_TWITTER_DATA_H
|
||||
|
||||
#include "twitter_data.h"
|
||||
#include <yyjson.h>
|
||||
#include <string>
|
||||
#include <stdexcept>
|
||||
|
||||
// yyjson deserialization for simplified Twitter data
|
||||
TwitterData yyjson_deserialize(const std::string &json_str) {
|
||||
TwitterData data;
|
||||
|
||||
yyjson_doc *doc = yyjson_read(json_str.c_str(), json_str.size(), 0);
|
||||
if (!doc) {
|
||||
throw std::runtime_error("yyjson parse error");
|
||||
}
|
||||
|
||||
yyjson_val *root = yyjson_doc_get_root(doc);
|
||||
if (!root) {
|
||||
yyjson_doc_free(doc);
|
||||
return data;
|
||||
}
|
||||
|
||||
// Get statuses array
|
||||
yyjson_val *statuses_val = yyjson_obj_get(root, "statuses");
|
||||
if (!statuses_val || !yyjson_is_arr(statuses_val)) {
|
||||
yyjson_doc_free(doc);
|
||||
return data;
|
||||
}
|
||||
|
||||
size_t idx, max;
|
||||
yyjson_val *status_val;
|
||||
yyjson_arr_foreach(statuses_val, idx, max, status_val) {
|
||||
Status status;
|
||||
|
||||
// Parse status fields
|
||||
yyjson_val *val;
|
||||
|
||||
val = yyjson_obj_get(status_val, "created_at");
|
||||
if (val && yyjson_is_str(val)) status.created_at = yyjson_get_str(val);
|
||||
|
||||
val = yyjson_obj_get(status_val, "id");
|
||||
if (val && yyjson_is_uint(val)) status.id = yyjson_get_uint(val);
|
||||
|
||||
val = yyjson_obj_get(status_val, "text");
|
||||
if (val && yyjson_is_str(val)) status.text = yyjson_get_str(val);
|
||||
|
||||
val = yyjson_obj_get(status_val, "retweet_count");
|
||||
if (val && yyjson_is_uint(val)) status.retweet_count = yyjson_get_uint(val);
|
||||
|
||||
val = yyjson_obj_get(status_val, "favorite_count");
|
||||
if (val && yyjson_is_uint(val)) status.favorite_count = yyjson_get_uint(val);
|
||||
|
||||
// Parse user
|
||||
yyjson_val *user_val = yyjson_obj_get(status_val, "user");
|
||||
if (user_val && yyjson_is_obj(user_val)) {
|
||||
User user;
|
||||
|
||||
val = yyjson_obj_get(user_val, "id");
|
||||
if (val && yyjson_is_uint(val)) user.id = yyjson_get_uint(val);
|
||||
|
||||
val = yyjson_obj_get(user_val, "name");
|
||||
if (val && yyjson_is_str(val)) user.name = yyjson_get_str(val);
|
||||
|
||||
val = yyjson_obj_get(user_val, "screen_name");
|
||||
if (val && yyjson_is_str(val)) user.screen_name = yyjson_get_str(val);
|
||||
|
||||
val = yyjson_obj_get(user_val, "location");
|
||||
if (val && yyjson_is_str(val)) user.location = yyjson_get_str(val);
|
||||
|
||||
val = yyjson_obj_get(user_val, "description");
|
||||
if (val && yyjson_is_str(val)) user.description = yyjson_get_str(val);
|
||||
|
||||
val = yyjson_obj_get(user_val, "verified");
|
||||
if (val && yyjson_is_bool(val)) user.verified = yyjson_get_bool(val);
|
||||
|
||||
val = yyjson_obj_get(user_val, "followers_count");
|
||||
if (val && yyjson_is_uint(val)) user.followers_count = yyjson_get_uint(val);
|
||||
|
||||
val = yyjson_obj_get(user_val, "friends_count");
|
||||
if (val && yyjson_is_uint(val)) user.friends_count = yyjson_get_uint(val);
|
||||
|
||||
val = yyjson_obj_get(user_val, "statuses_count");
|
||||
if (val && yyjson_is_uint(val)) user.statuses_count = yyjson_get_uint(val);
|
||||
|
||||
status.user = user;
|
||||
}
|
||||
|
||||
data.statuses.push_back(status);
|
||||
}
|
||||
|
||||
yyjson_doc_free(doc);
|
||||
return data;
|
||||
}
|
||||
|
||||
// yyjson serialization for simplified Twitter data
|
||||
std::string yyjson_serialize(const TwitterData &data) {
|
||||
yyjson_mut_doc *doc = yyjson_mut_doc_new(NULL);
|
||||
yyjson_mut_val *root = yyjson_mut_obj(doc);
|
||||
yyjson_mut_doc_set_root(doc, root);
|
||||
|
||||
// Create statuses array
|
||||
yyjson_mut_val *statuses_array = yyjson_mut_arr(doc);
|
||||
|
||||
for (const auto& status : data.statuses) {
|
||||
yyjson_mut_val *status_obj = yyjson_mut_obj(doc);
|
||||
|
||||
// Add status fields
|
||||
yyjson_mut_obj_add_str(doc, status_obj, "created_at", status.created_at.c_str());
|
||||
yyjson_mut_obj_add_uint(doc, status_obj, "id", status.id);
|
||||
yyjson_mut_obj_add_str(doc, status_obj, "text", status.text.c_str());
|
||||
|
||||
// User object
|
||||
yyjson_mut_val *user_obj = yyjson_mut_obj(doc);
|
||||
yyjson_mut_obj_add_uint(doc, user_obj, "id", status.user.id);
|
||||
yyjson_mut_obj_add_str(doc, user_obj, "name", status.user.name.c_str());
|
||||
yyjson_mut_obj_add_str(doc, user_obj, "screen_name", status.user.screen_name.c_str());
|
||||
yyjson_mut_obj_add_str(doc, user_obj, "location", status.user.location.c_str());
|
||||
yyjson_mut_obj_add_str(doc, user_obj, "description", status.user.description.c_str());
|
||||
yyjson_mut_obj_add_bool(doc, user_obj, "verified", status.user.verified);
|
||||
yyjson_mut_obj_add_uint(doc, user_obj, "followers_count", status.user.followers_count);
|
||||
yyjson_mut_obj_add_uint(doc, user_obj, "friends_count", status.user.friends_count);
|
||||
yyjson_mut_obj_add_uint(doc, user_obj, "statuses_count", status.user.statuses_count);
|
||||
yyjson_mut_obj_add_val(doc, status_obj, "user", user_obj);
|
||||
|
||||
// Other fields
|
||||
yyjson_mut_obj_add_uint(doc, status_obj, "retweet_count", status.retweet_count);
|
||||
yyjson_mut_obj_add_uint(doc, status_obj, "favorite_count", status.favorite_count);
|
||||
|
||||
yyjson_mut_arr_append(statuses_array, status_obj);
|
||||
}
|
||||
|
||||
// Add statuses array to root
|
||||
yyjson_mut_obj_add_val(doc, root, "statuses", statuses_array);
|
||||
|
||||
// Write to string
|
||||
char *json_output = yyjson_mut_write(doc, 0, NULL);
|
||||
std::string result(json_output);
|
||||
free(json_output);
|
||||
yyjson_mut_doc_free(doc);
|
||||
|
||||
return result;
|
||||
}
|
||||
|
||||
#endif // YYJSON_TWITTER_DATA_H
|
||||
+1
-1
@@ -469,7 +469,7 @@ support for users who avoid exceptions. See [the simdjson error handling documen
|
||||
out of the array, you may use an array access (e.g., `array[1]`). You should never reset an array as you are iterating through it. The following is an anti-pattern: `for(auto value: myarray) {myarray.reset()}`.
|
||||
* **Field Access:** To get the value of the "foo" field in an object, use `object["foo"]`. This will
|
||||
scan through the object looking for the field with the matching string, doing a character-by-character
|
||||
comparison. It may generate the error `simdjson::NO_SUCH_FIELD` if there is no such key in the object, it may throw an exception (see [Error handling](#error-handling)). For efficiency reason, you should avoid looking up the same field repeatedly: e.g., do
|
||||
comparison. It may generate the error `simdjson::NO_SUCH_FIELD` if there is no such key in the object, it may throw an exception (see [Error handling](#error-handling)). The returned value is only valid so long as you do not access another field: normally, you should therefore grab the value right after accessing a key (i.e., convert it to number, string, object, array...). For efficiency reason, you should avoid looking up the same field repeatedly: e.g., do
|
||||
not do `object["foo"]` followed by `object["foo"]` with the same `object` instance. Generally, you should not mix and match iterating through an object (`for(auto field : object) {...}`) and key accesses (`object["foo"]`): if you need to iterate through an object after a key access, you need to call `reset()` on the object. Whenever you call `reset()`, you need to keep in mind that though you can iterate over the array repeatedly, values should be consumedonly once (e.g., repeatedly calling `unescaped_key()` on the same key is forbidden). Keep in mind that On-Demand does not buffer or save the result of the parsing: if you repeatedly access `object["foo"]`, then it must repeatedly seek the key and parse the content. The library does not provide a distinct function to check if a key is present, instead we recommend you attempt to access the key: e.g., by doing `ondemand::value val{}; if (!object["foo"].get(val)) {...}`, you have that `val` contains the requested value inside the if clause. It is your responsibility as a user to temporarily keep a reference to the value (`auto v = object["foo"]`), or to consume the content and store it in your own data structures. If you consume an
|
||||
object twice: `std::string_view(object["foo"]` followed by `std::string_view(object["foo"]` then your code
|
||||
is in error. Furthermore, you can only consume one field at a time, on the same object. The
|
||||
|
||||
@@ -428,6 +428,7 @@ namespace {
|
||||
static constexpr int NUM_CHUNKS = 64 / sizeof(simd8<T>);
|
||||
static_assert(NUM_CHUNKS == 4, "ARM kernel should use four registers per 64-byte block.");
|
||||
const simd8<T> chunks[NUM_CHUNKS];
|
||||
template<int idx> simd8<uint8_t> get() const { return idx < NUM_CHUNKS ? chunks[idx] : simd8<T>(); }
|
||||
|
||||
simd8x64(const simd8x64<T>& o) = delete; // no copy allowed
|
||||
simd8x64<T>& operator=(const simd8<T>& other) = delete; // no assignment allowed
|
||||
|
||||
@@ -24,6 +24,8 @@
|
||||
#include "simdjson/lasx.h"
|
||||
#elif SIMDJSON_BUILTIN_IMPLEMENTATION_IS(lsx)
|
||||
#include "simdjson/lsx.h"
|
||||
#elif SIMDJSON_BUILTIN_IMPLEMENTATION_IS(rvv_vls)
|
||||
#include "simdjson/rvv-vls.h"
|
||||
#else
|
||||
#error Unknown SIMDJSON_BUILTIN_IMPLEMENTATION
|
||||
#endif
|
||||
|
||||
@@ -21,6 +21,8 @@ namespace simdjson {
|
||||
namespace lsx {}
|
||||
#elif SIMDJSON_BUILTIN_IMPLEMENTATION_IS(lasx)
|
||||
namespace lasx {}
|
||||
#elif SIMDJSON_BUILTIN_IMPLEMENTATION_IS(rvv_vls)
|
||||
namespace rvv_vls {}
|
||||
#else
|
||||
#error Unknown SIMDJSON_BUILTIN_IMPLEMENTATION
|
||||
#endif
|
||||
|
||||
@@ -24,6 +24,8 @@
|
||||
#include "simdjson/lsx/builder.h"
|
||||
#elif SIMDJSON_BUILTIN_IMPLEMENTATION_IS(lasx)
|
||||
#include "simdjson/lasx/builder.h"
|
||||
#elif SIMDJSON_BUILTIN_IMPLEMENTATION_IS(rvv_vls)
|
||||
#include "simdjson/rvv-vls/builder.h"
|
||||
#else
|
||||
#error Unknown SIMDJSON_BUILTIN_IMPLEMENTATION
|
||||
#endif
|
||||
@@ -37,4 +39,4 @@ namespace simdjson {
|
||||
namespace builder = SIMDJSON_BUILTIN_IMPLEMENTATION::builder;
|
||||
} // namespace simdjson
|
||||
|
||||
#endif // SIMDJSON_BUILTIN_BUILDER_H
|
||||
#endif // SIMDJSON_BUILTIN_BUILDER_H
|
||||
|
||||
@@ -23,6 +23,8 @@
|
||||
#include "simdjson/lsx/implementation.h"
|
||||
#elif SIMDJSON_BUILTIN_IMPLEMENTATION_IS(lasx)
|
||||
#include "simdjson/lasx/implementation.h"
|
||||
#elif SIMDJSON_BUILTIN_IMPLEMENTATION_IS(rvv_vls)
|
||||
#include "simdjson/rvv-vls/implementation.h"
|
||||
#else
|
||||
#error Unknown SIMDJSON_BUILTIN_IMPLEMENTATION
|
||||
#endif
|
||||
@@ -39,4 +41,4 @@ namespace simdjson {
|
||||
const implementation * builtin_implementation();
|
||||
} // namespace simdjson
|
||||
|
||||
#endif // SIMDJSON_BUILTIN_IMPLEMENTATION_H
|
||||
#endif // SIMDJSON_BUILTIN_IMPLEMENTATION_H
|
||||
|
||||
@@ -24,6 +24,8 @@
|
||||
#include "simdjson/lsx/ondemand.h"
|
||||
#elif SIMDJSON_BUILTIN_IMPLEMENTATION_IS(lasx)
|
||||
#include "simdjson/lasx/ondemand.h"
|
||||
#elif SIMDJSON_BUILTIN_IMPLEMENTATION_IS(rvv_vls)
|
||||
#include "simdjson/rvv-vls/ondemand.h"
|
||||
#else
|
||||
#error Unknown SIMDJSON_BUILTIN_IMPLEMENTATION
|
||||
#endif
|
||||
@@ -37,4 +39,4 @@ namespace simdjson {
|
||||
namespace ondemand = SIMDJSON_BUILTIN_IMPLEMENTATION::ondemand;
|
||||
} // namespace simdjson
|
||||
|
||||
#endif // SIMDJSON_BUILTIN_ONDEMAND_H
|
||||
#endif // SIMDJSON_BUILTIN_ONDEMAND_H
|
||||
|
||||
@@ -21,6 +21,8 @@
|
||||
#include "simdjson/lasx/begin.h"
|
||||
#elif SIMDJSON_IMPLEMENTATION_LSX
|
||||
#include "simdjson/lsx/begin.h"
|
||||
#elif SIMDJSON_IMPLEMENTATION_RVV_VLS
|
||||
#include "simdjson/rvv-vls/begin.h"
|
||||
#elif SIMDJSON_IMPLEMENTATION_FALLBACK
|
||||
#include "simdjson/fallback/begin.h"
|
||||
#else
|
||||
@@ -48,4 +50,4 @@ enum class number_type {
|
||||
} // namespace SIMDJSON_IMPLEMENTATION
|
||||
} // namespace simdjson
|
||||
|
||||
#endif // SIMDJSON_GENERIC_BASE_H
|
||||
#endif // SIMDJSON_GENERIC_BASE_H
|
||||
|
||||
@@ -275,7 +275,7 @@ simdjson_warn_unused simdjson_result<std::string> to_json_string(const Z &z, siz
|
||||
}
|
||||
|
||||
template <class Z>
|
||||
simdjson_warn_unused simdjson_error to_json(const Z &z, std::string &s, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
|
||||
simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
|
||||
string_builder b(initial_capacity);
|
||||
append(b, z);
|
||||
std::string_view view;
|
||||
@@ -352,7 +352,7 @@ simdjson_warn_unused simdjson_result<std::string> to_json(const Z &z, size_t ini
|
||||
return std::string(s);
|
||||
}
|
||||
template <class Z>
|
||||
simdjson_warn_unused simdjson_error to_json(const Z &z, std::string &s, size_t initial_capacity = SIMDJSON_IMPLEMENTATION::builder::string_builder::DEFAULT_INITIAL_CAPACITY) {
|
||||
simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initial_capacity = SIMDJSON_IMPLEMENTATION::builder::string_builder::DEFAULT_INITIAL_CAPACITY) {
|
||||
SIMDJSON_IMPLEMENTATION::builder::string_builder b(initial_capacity);
|
||||
SIMDJSON_IMPLEMENTATION::builder::append(b, z);
|
||||
std::string_view view;
|
||||
|
||||
@@ -29,9 +29,15 @@
|
||||
#endif
|
||||
#if SIMDJSON_EXPERIMENTAL_HAS_NEON
|
||||
#include <arm_neon.h>
|
||||
#ifdef _MSC_VER
|
||||
#include <intrin.h>
|
||||
#endif
|
||||
#endif
|
||||
#if SIMDJSON_EXPERIMENTAL_HAS_SSE2
|
||||
#include <emmintrin.h>
|
||||
#ifdef _MSC_VER
|
||||
#include <intrin.h>
|
||||
#endif
|
||||
#endif
|
||||
|
||||
namespace simdjson {
|
||||
@@ -139,10 +145,10 @@ simdjson_inline bool fast_needs_escaping(std::string_view view) {
|
||||
}
|
||||
#endif
|
||||
|
||||
SIMDJSON_CONSTEXPR_LAMBDA inline size_t
|
||||
find_next_json_quotable_character(const std::string_view view,
|
||||
size_t location) noexcept {
|
||||
|
||||
// Scalar fallback for finding next quotable character
|
||||
SIMDJSON_CONSTEXPR_LAMBDA simdjson_inline size_t
|
||||
find_next_json_quotable_character_scalar(const std::string_view view,
|
||||
size_t location) noexcept {
|
||||
for (auto pos = view.begin() + location; pos != view.end(); ++pos) {
|
||||
if (json_quotable_character[static_cast<uint8_t>(*pos)]) {
|
||||
return pos - view.begin();
|
||||
@@ -151,6 +157,114 @@ find_next_json_quotable_character(const std::string_view view,
|
||||
return size_t(view.size());
|
||||
}
|
||||
|
||||
// SIMD-accelerated position finding that directly locates the first quotable
|
||||
// character, combining detection and position extraction in a single pass to
|
||||
// minimize redundant work.
|
||||
#if SIMDJSON_EXPERIMENTAL_HAS_NEON
|
||||
simdjson_inline size_t
|
||||
find_next_json_quotable_character(const std::string_view view,
|
||||
size_t location) noexcept {
|
||||
const size_t len = view.size();
|
||||
const uint8_t *ptr =
|
||||
reinterpret_cast<const uint8_t *>(view.data()) + location;
|
||||
size_t remaining = len - location;
|
||||
|
||||
// SIMD constants for characters requiring escape
|
||||
uint8x16_t v34 = vdupq_n_u8(34); // '"'
|
||||
uint8x16_t v92 = vdupq_n_u8(92); // '\\'
|
||||
uint8x16_t v32 = vdupq_n_u8(32); // control char threshold
|
||||
|
||||
while (remaining >= 16) {
|
||||
uint8x16_t word = vld1q_u8(ptr);
|
||||
|
||||
// Check for quotable characters: '"', '\\', or control chars (< 32)
|
||||
uint8x16_t needs_escape = vceqq_u8(word, v34);
|
||||
needs_escape = vorrq_u8(needs_escape, vceqq_u8(word, v92));
|
||||
needs_escape = vorrq_u8(needs_escape, vcltq_u8(word, v32));
|
||||
|
||||
if (vmaxvq_u32(vreinterpretq_u32_u8(needs_escape)) != 0) {
|
||||
// Found quotable character - extract exact byte position using ctz
|
||||
uint64x2_t as64 = vreinterpretq_u64_u8(needs_escape);
|
||||
uint64_t lo = vgetq_lane_u64(as64, 0);
|
||||
uint64_t hi = vgetq_lane_u64(as64, 1);
|
||||
size_t offset = ptr - reinterpret_cast<const uint8_t *>(view.data());
|
||||
#ifdef _MSC_VER
|
||||
unsigned long trailing_zero = 0;
|
||||
if (lo != 0) {
|
||||
_BitScanForward64(&trailing_zero, lo);
|
||||
return offset + trailing_zero / 8;
|
||||
} else {
|
||||
_BitScanForward64(&trailing_zero, hi);
|
||||
return offset + 8 + trailing_zero / 8;
|
||||
}
|
||||
#else
|
||||
if (lo != 0) {
|
||||
return offset + __builtin_ctzll(lo) / 8;
|
||||
} else {
|
||||
return offset + 8 + __builtin_ctzll(hi) / 8;
|
||||
}
|
||||
#endif
|
||||
}
|
||||
ptr += 16;
|
||||
remaining -= 16;
|
||||
}
|
||||
|
||||
// Scalar fallback for remaining bytes
|
||||
size_t current = len - remaining;
|
||||
return find_next_json_quotable_character_scalar(view, current);
|
||||
}
|
||||
#elif SIMDJSON_EXPERIMENTAL_HAS_SSE2
|
||||
simdjson_inline size_t
|
||||
find_next_json_quotable_character(const std::string_view view,
|
||||
size_t location) noexcept {
|
||||
const size_t len = view.size();
|
||||
const uint8_t *ptr =
|
||||
reinterpret_cast<const uint8_t *>(view.data()) + location;
|
||||
size_t remaining = len - location;
|
||||
|
||||
// SIMD constants
|
||||
__m128i v34 = _mm_set1_epi8(34); // '"'
|
||||
__m128i v92 = _mm_set1_epi8(92); // '\\'
|
||||
__m128i v31 = _mm_set1_epi8(31); // for control char detection
|
||||
|
||||
while (remaining >= 16) {
|
||||
__m128i word = _mm_loadu_si128(reinterpret_cast<const __m128i *>(ptr));
|
||||
|
||||
// Check for quotable characters
|
||||
__m128i needs_escape = _mm_cmpeq_epi8(word, v34);
|
||||
needs_escape = _mm_or_si128(needs_escape, _mm_cmpeq_epi8(word, v92));
|
||||
needs_escape = _mm_or_si128(
|
||||
needs_escape,
|
||||
_mm_cmpeq_epi8(_mm_subs_epu8(word, v31), _mm_setzero_si128()));
|
||||
|
||||
int mask = _mm_movemask_epi8(needs_escape);
|
||||
if (mask != 0) {
|
||||
// Found quotable character - use trailing zero count to find position
|
||||
size_t offset = ptr - reinterpret_cast<const uint8_t *>(view.data());
|
||||
#ifdef _MSC_VER
|
||||
unsigned long trailing_zero = 0;
|
||||
_BitScanForward(&trailing_zero, mask);
|
||||
return offset + trailing_zero;
|
||||
#else
|
||||
return offset + __builtin_ctz(mask);
|
||||
#endif
|
||||
}
|
||||
ptr += 16;
|
||||
remaining -= 16;
|
||||
}
|
||||
|
||||
// Scalar fallback for remaining bytes
|
||||
size_t current = len - remaining;
|
||||
return find_next_json_quotable_character_scalar(view, current);
|
||||
}
|
||||
#else
|
||||
SIMDJSON_CONSTEXPR_LAMBDA simdjson_inline size_t
|
||||
find_next_json_quotable_character(const std::string_view view,
|
||||
size_t location) noexcept {
|
||||
return find_next_json_quotable_character_scalar(view, location);
|
||||
}
|
||||
#endif
|
||||
|
||||
SIMDJSON_CONSTEXPR_LAMBDA static std::string_view control_chars[] = {
|
||||
"\\u0000", "\\u0001", "\\u0002", "\\u0003", "\\u0004", "\\u0005", "\\u0006",
|
||||
"\\u0007", "\\b", "\\t", "\\n", "\\u000b", "\\f", "\\r",
|
||||
@@ -163,7 +277,7 @@ SIMDJSON_CONSTEXPR_LAMBDA static std::string_view control_chars[] = {
|
||||
// control characters (U+0000 through U+001F). There are two-character sequence
|
||||
// escape representations of some popular characters:
|
||||
// \", \\, \b, \f, \n, \r, \t.
|
||||
SIMDJSON_CONSTEXPR_LAMBDA void escape_json_char(char c, char *&out) {
|
||||
SIMDJSON_CONSTEXPR_LAMBDA simdjson_inline void escape_json_char(char c, char *&out) {
|
||||
if (c == '"') {
|
||||
memcpy(out, "\\\"", 2);
|
||||
out += 2;
|
||||
@@ -177,14 +291,20 @@ SIMDJSON_CONSTEXPR_LAMBDA void escape_json_char(char c, char *&out) {
|
||||
}
|
||||
}
|
||||
|
||||
// Writes the escaped version of input to out, returning the number of bytes
|
||||
// written. Uses SIMD position finding to locate quotable characters efficiently.
|
||||
inline size_t write_string_escaped(const std::string_view input, char *out) {
|
||||
size_t mysize = input.size();
|
||||
if (!fast_needs_escaping(input)) { // fast path!
|
||||
|
||||
// Use SIMD position finder directly - it returns mysize if no escape needed
|
||||
size_t location = find_next_json_quotable_character(input, 0);
|
||||
if (location == mysize) {
|
||||
// Fast path: no escaping needed
|
||||
memcpy(out, input.data(), input.size());
|
||||
return input.size();
|
||||
}
|
||||
|
||||
const char *const initout = out;
|
||||
size_t location = find_next_json_quotable_character(input, 0);
|
||||
memcpy(out, input.data(), location);
|
||||
out += location;
|
||||
escape_json_char(input[location], out);
|
||||
@@ -383,12 +503,29 @@ simdjson_inline void string_builder::append(number_type v) noexcept {
|
||||
}
|
||||
}
|
||||
else SIMDJSON_IF_CONSTEXPR(std::is_unsigned<number_type>::value) {
|
||||
// Process 4 digits at a time instead of 2, reducing store operations
|
||||
// and divisions by approximately half for large numbers.
|
||||
constexpr size_t max_number_size = 20;
|
||||
if (capacity_check(max_number_size)) {
|
||||
using unsigned_type = typename std::make_unsigned<number_type>::type;
|
||||
unsigned_type pv = static_cast<unsigned_type>(v);
|
||||
size_t dc = internal::digit_count(pv);
|
||||
char *write_pointer = buffer.get() + position + dc - 1;
|
||||
|
||||
// Process 4 digits per iteration for large numbers
|
||||
while (pv >= 10000) {
|
||||
unsigned_type q = pv / 10000;
|
||||
unsigned_type r = pv % 10000;
|
||||
unsigned_type r_hi = r / 100; // High 2 digits of remainder
|
||||
unsigned_type r_lo = r % 100; // Low 2 digits of remainder
|
||||
// Write low 2 digits first (rightmost), then high 2 digits
|
||||
memcpy(write_pointer - 1, &internal::decimal_table[r_lo * 2], 2);
|
||||
memcpy(write_pointer - 3, &internal::decimal_table[r_hi * 2], 2);
|
||||
write_pointer -= 4;
|
||||
pv = q;
|
||||
}
|
||||
|
||||
// Handle remaining 1-4 digits with original 2-digit loop
|
||||
while (pv >= 100) {
|
||||
memcpy(write_pointer - 1, &internal::decimal_table[(pv % 100) * 2], 2);
|
||||
write_pointer -= 2;
|
||||
@@ -403,6 +540,7 @@ simdjson_inline void string_builder::append(number_type v) noexcept {
|
||||
}
|
||||
}
|
||||
else SIMDJSON_IF_CONSTEXPR(std::is_integral<number_type>::value) {
|
||||
// Same 4-digit batching as unsigned path for signed integers
|
||||
constexpr size_t max_number_size = 20;
|
||||
if (capacity_check(max_number_size)) {
|
||||
using unsigned_type = typename std::make_unsigned<number_type>::type;
|
||||
@@ -416,6 +554,20 @@ simdjson_inline void string_builder::append(number_type v) noexcept {
|
||||
buffer.get()[position] = '-';
|
||||
position += negative ? 1 : 0;
|
||||
char *write_pointer = buffer.get() + position + dc - 1;
|
||||
|
||||
// Process 4 digits per iteration for large numbers
|
||||
while (pv >= 10000) {
|
||||
unsigned_type q = pv / 10000;
|
||||
unsigned_type r = pv % 10000;
|
||||
unsigned_type r_hi = r / 100;
|
||||
unsigned_type r_lo = r % 100;
|
||||
memcpy(write_pointer - 1, &internal::decimal_table[r_lo * 2], 2);
|
||||
memcpy(write_pointer - 3, &internal::decimal_table[r_hi * 2], 2);
|
||||
write_pointer -= 4;
|
||||
pv = q;
|
||||
}
|
||||
|
||||
// Handle remaining 1-4 digits
|
||||
while (pv >= 100) {
|
||||
memcpy(write_pointer - 1, &internal::decimal_table[(pv % 100) * 2], 2);
|
||||
write_pointer -= 2;
|
||||
|
||||
@@ -280,7 +280,7 @@ simdjson_warn_unused simdjson_result<std::string> to_json(const Z &z, size_t ini
|
||||
return std::string(s);
|
||||
}
|
||||
template <class Z>
|
||||
simdjson_warn_unused simdjson_error to_json(const Z &z, std::string &s, size_t initial_capacity = simdjson::SIMDJSON_IMPLEMENTATION::builder::string_builder::DEFAULT_INITIAL_CAPACITY) {
|
||||
simdjson_warn_unused error_code to_json(const Z &z, std::string &s, size_t initial_capacity = simdjson::SIMDJSON_IMPLEMENTATION::builder::string_builder::DEFAULT_INITIAL_CAPACITY) {
|
||||
simdjson::SIMDJSON_IMPLEMENTATION::builder::string_builder b(initial_capacity);
|
||||
b.append(z);
|
||||
std::string_view sv;
|
||||
|
||||
@@ -515,9 +515,13 @@ simdjson_warn_unused simdjson_inline simdjson_result<std::string_view> value_ite
|
||||
template <typename string_type>
|
||||
simdjson_warn_unused simdjson_inline error_code value_iterator::get_string(string_type& receiver, bool allow_replacement) noexcept {
|
||||
std::string_view content;
|
||||
// Save the string buffer location so that we can restore it after get_string
|
||||
auto saved_string_buf_loc = _json_iter->string_buf_loc();
|
||||
auto err = get_string(allow_replacement).get(content);
|
||||
if (err) { return err; }
|
||||
receiver = content;
|
||||
// Restore the string buffer location, effectively discarding any temporary string storage
|
||||
_json_iter->string_buf_loc() = saved_string_buf_loc;
|
||||
return SUCCESS;
|
||||
}
|
||||
simdjson_warn_unused simdjson_inline simdjson_result<std::string_view> value_iterator::get_wobbly_string() noexcept {
|
||||
@@ -657,9 +661,13 @@ simdjson_warn_unused simdjson_inline simdjson_result<std::string_view> value_ite
|
||||
template <typename string_type>
|
||||
simdjson_warn_unused simdjson_inline error_code value_iterator::get_root_string(string_type& receiver, bool check_trailing, bool allow_replacement) noexcept {
|
||||
std::string_view content;
|
||||
// Save the string buffer location so that we can restore it after get_string
|
||||
auto saved_string_buf_loc = _json_iter->string_buf_loc();
|
||||
auto err = get_root_string(check_trailing, allow_replacement).get(content);
|
||||
if (err) { return err; }
|
||||
receiver = content;
|
||||
// Restore the string buffer location, effectively discarding any temporary string storage
|
||||
_json_iter->string_buf_loc() = saved_string_buf_loc;
|
||||
return SUCCESS;
|
||||
}
|
||||
simdjson_warn_unused simdjson_inline simdjson_result<std::string_view> value_iterator::get_root_wobbly_string(bool check_trailing) noexcept {
|
||||
|
||||
@@ -300,6 +300,7 @@ namespace simd {
|
||||
static constexpr int NUM_CHUNKS = 64 / sizeof(simd8<T>);
|
||||
static_assert(NUM_CHUNKS == 2, "Haswell kernel should use two registers per 64-byte block.");
|
||||
const simd8<T> chunks[NUM_CHUNKS];
|
||||
template<int idx> simd8<uint8_t> get() const { return idx < NUM_CHUNKS ? chunks[idx] : simd8<T>(); }
|
||||
|
||||
simd8x64(const simd8x64<T>& o) = delete; // no copy allowed
|
||||
simd8x64<T>& operator=(const simd8<T>& other) = delete; // no assignment allowed
|
||||
|
||||
@@ -322,6 +322,7 @@ namespace simd {
|
||||
static constexpr int NUM_CHUNKS = 64 / sizeof(simd8<T>);
|
||||
static_assert(NUM_CHUNKS == 1, "Icelake kernel should use one register per 64-byte block.");
|
||||
const simd8<T> chunks[NUM_CHUNKS];
|
||||
template<int idx> simd8<uint8_t> get() const { return idx < NUM_CHUNKS ? chunks[idx] : simd8<T>(); }
|
||||
|
||||
simd8x64(const simd8x64<T>& o) = delete; // no copy allowed
|
||||
simd8x64<T>& operator=(const simd8<T>& other) = delete; // no assignment allowed
|
||||
|
||||
@@ -12,6 +12,8 @@
|
||||
#define SIMDJSON_IMPLEMENTATION_ID_westmere 6
|
||||
#define SIMDJSON_IMPLEMENTATION_ID_lsx 7
|
||||
#define SIMDJSON_IMPLEMENTATION_ID_lasx 8
|
||||
//#define SIMDJSON_IMPLEMENTATION_ID_rvv 9
|
||||
#define SIMDJSON_IMPLEMENTATION_ID_rvv_vls 10
|
||||
|
||||
#define SIMDJSON_IMPLEMENTATION_ID_FOR(IMPL) SIMDJSON_CAT(SIMDJSON_IMPLEMENTATION_ID_, IMPL)
|
||||
#define SIMDJSON_IMPLEMENTATION_ID SIMDJSON_IMPLEMENTATION_ID_FOR(SIMDJSON_IMPLEMENTATION)
|
||||
@@ -126,9 +128,14 @@
|
||||
#endif
|
||||
#define SIMDJSON_CAN_ALWAYS_RUN_LSX (SIMDJSON_IMPLEMENTATION_LSX)
|
||||
|
||||
#define SIMDJSON_CAN_ALWAYS_RUN_RVV_VLS SIMDJSON_IS_RVV_VLS
|
||||
#ifndef SIMDJSON_IMPLEMENTATION_RVV_VLS
|
||||
#define SIMDJSON_IMPLEMENTATION_RVV_VLS SIMDJSON_CAN_ALWAYS_RUN_RVV_VLS
|
||||
#endif
|
||||
|
||||
// Default Fallback to on unless a builtin implementation has already been selected.
|
||||
#ifndef SIMDJSON_IMPLEMENTATION_FALLBACK
|
||||
#if SIMDJSON_CAN_ALWAYS_RUN_ARM64 || SIMDJSON_CAN_ALWAYS_RUN_ICELAKE || SIMDJSON_CAN_ALWAYS_RUN_HASWELL || SIMDJSON_CAN_ALWAYS_RUN_WESTMERE || SIMDJSON_CAN_ALWAYS_RUN_PPC64 || SIMDJSON_CAN_ALWAYS_RUN_LSX || SIMDJSON_CAN_ALWAYS_RUN_LASX
|
||||
#if SIMDJSON_CAN_ALWAYS_RUN_ARM64 || SIMDJSON_CAN_ALWAYS_RUN_ICELAKE || SIMDJSON_CAN_ALWAYS_RUN_HASWELL || SIMDJSON_CAN_ALWAYS_RUN_WESTMERE || SIMDJSON_CAN_ALWAYS_RUN_PPC64 || SIMDJSON_CAN_ALWAYS_RUN_LSX || SIMDJSON_CAN_ALWAYS_RUN_LASX || SIMDJSON_CAN_ALWAYS_RUN_RVV_VLS
|
||||
// if anything at all except fallback can always run, then disable fallback.
|
||||
#define SIMDJSON_IMPLEMENTATION_FALLBACK 0
|
||||
#else
|
||||
@@ -154,6 +161,8 @@
|
||||
#define SIMDJSON_BUILTIN_IMPLEMENTATION lsx
|
||||
#elif SIMDJSON_CAN_ALWAYS_RUN_LASX
|
||||
#define SIMDJSON_BUILTIN_IMPLEMENTATION lasx
|
||||
#elif SIMDJSON_CAN_ALWAYS_RUN_RVV_VLS
|
||||
#define SIMDJSON_BUILTIN_IMPLEMENTATION rvv_vls
|
||||
#elif SIMDJSON_CAN_ALWAYS_RUN_FALLBACK
|
||||
#define SIMDJSON_BUILTIN_IMPLEMENTATION fallback
|
||||
#else
|
||||
|
||||
@@ -69,6 +69,8 @@ enum instruction_set {
|
||||
AVX512VBMI2 = 0x10000,
|
||||
LSX = 0x20000,
|
||||
LASX = 0x40000,
|
||||
//RVV = 0x80000,
|
||||
RVV_VLS = 0x100000,
|
||||
};
|
||||
|
||||
} // namespace internal
|
||||
|
||||
@@ -303,6 +303,7 @@ namespace simd {
|
||||
static constexpr int NUM_CHUNKS = 64 / sizeof(simd8<T>);
|
||||
static_assert(NUM_CHUNKS == 2, "LASX kernel should use two registers per 64-byte block.");
|
||||
const simd8<T> chunks[NUM_CHUNKS];
|
||||
template<int idx> simd8<uint8_t> get() const { return idx < NUM_CHUNKS ? chunks[idx] : simd8<T>(); }
|
||||
|
||||
simd8x64(const simd8x64<T>& o) = delete; // no copy allowed
|
||||
simd8x64<T>& operator=(const simd8<T>& other) = delete; // no assignment allowed
|
||||
|
||||
@@ -260,6 +260,7 @@ namespace simd {
|
||||
static constexpr int NUM_CHUNKS = 64 / sizeof(simd8<T>);
|
||||
static_assert(NUM_CHUNKS == 4, "LSX kernel should use four registers per 64-byte block.");
|
||||
const simd8<T> chunks[NUM_CHUNKS];
|
||||
template<int idx> simd8<uint8_t> get() const { return idx < NUM_CHUNKS ? chunks[idx] : simd8<T>(); }
|
||||
|
||||
simd8x64(const simd8x64<T>& o) = delete; // no copy allowed
|
||||
simd8x64<T>& operator=(const simd8<T>& other) = delete; // no assignment allowed
|
||||
|
||||
@@ -45,22 +45,20 @@ using std::size_t;
|
||||
#define SIMDJSON_IS_ARM64 1
|
||||
#elif defined(__riscv) && __riscv_xlen == 64
|
||||
#define SIMDJSON_IS_RISCV64 1
|
||||
|
||||
#if __riscv_v_intrinsic >= 11000
|
||||
#define SIMDJSON_HAS_RVV_INTRINSICS 1
|
||||
#endif
|
||||
|
||||
#define SIMDJSON_HAS_ZVBB_INTRINSICS \
|
||||
0 // there is currently no way to detect this
|
||||
|
||||
#if SIMDJSON_HAS_RVV_INTRINSICS && __riscv_vector && \
|
||||
__riscv_v_min_vlen >= 128 && __riscv_v_elen >= 64
|
||||
// RISC-V V extension
|
||||
#define SIMDJSON_IS_RVV 1
|
||||
#if SIMDJSON_HAS_ZVBB_INTRINSICS && __riscv_zvbb >= 1000000
|
||||
// RISC-V Vector Basic Bit-manipulation
|
||||
#define SIMDJSON_IS_ZVBB 1
|
||||
#endif
|
||||
#if SIMDJSON_HAS_RVV_INTRINSICS && __riscv_vector && __riscv_v_min_vlen >= 128 && __riscv_v_elen >= 64
|
||||
#define SIMDJSON_IS_RVV 1 // RISC-V V extension
|
||||
#endif
|
||||
|
||||
// current toolchains don't support fixed-size SIMD types that don't match VLEN directly
|
||||
#if __riscv_v_fixed_vlen >= 128 && __riscv_v_fixed_vlen <= 512
|
||||
#define SIMDJSON_IS_RVV_VLS 1
|
||||
#endif
|
||||
|
||||
#elif defined(__loongarch_lp64)
|
||||
#define SIMDJSON_IS_LOONGARCH64 1
|
||||
#if defined(__loongarch_sx) && defined(__loongarch_asx)
|
||||
|
||||
@@ -397,6 +397,7 @@ template <typename T> struct simd8x64 {
|
||||
static_assert(NUM_CHUNKS == 4,
|
||||
"PPC64 kernel should use four registers per 64-byte block.");
|
||||
const simd8<T> chunks[NUM_CHUNKS];
|
||||
template<int idx> simd8<uint8_t> get() const { return idx < NUM_CHUNKS ? chunks[idx] : simd8<T>(); }
|
||||
|
||||
simd8x64(const simd8x64<T> &o) = delete; // no copy allowed
|
||||
simd8x64<T> &
|
||||
|
||||
@@ -0,0 +1,9 @@
|
||||
#ifndef SIMDJSON_RVV_VLS_H
|
||||
#define SIMDJSON_RVV_VLS_H
|
||||
|
||||
|
||||
#include "simdjson/rvv-vls/begin.h"
|
||||
#include "simdjson/generic/amalgamated.h"
|
||||
#include "simdjson/rvv-vls/end.h"
|
||||
|
||||
#endif // SIMDJSON_RVV_VLS_H
|
||||
@@ -0,0 +1,19 @@
|
||||
#ifndef SIMDJSON_RVV_VLS_BASE_H
|
||||
#define SIMDJSON_RVV_VLS_BASE_H
|
||||
|
||||
#ifndef SIMDJSON_CONDITIONAL_INCLUDE
|
||||
#include "simdjson/base.h"
|
||||
#endif // SIMDJSON_CONDITIONAL_INCLUDE
|
||||
|
||||
namespace simdjson {
|
||||
/**
|
||||
* RVV-VLS implementation.
|
||||
*/
|
||||
namespace rvv_vls {
|
||||
|
||||
class implementation;
|
||||
|
||||
} // namespace rvv_vls
|
||||
} // namespace simdjson
|
||||
|
||||
#endif // SIMDJSON_RVV_VLS_BASE_H
|
||||
@@ -0,0 +1,10 @@
|
||||
#define SIMDJSON_IMPLEMENTATION rvv_vls
|
||||
#include "simdjson/rvv-vls/base.h"
|
||||
#include "simdjson/rvv-vls/intrinsics.h"
|
||||
#include "simdjson/rvv-vls/bitmanipulation.h"
|
||||
#include "simdjson/rvv-vls/bitmask.h"
|
||||
#include "simdjson/rvv-vls/simd.h"
|
||||
#include "simdjson/rvv-vls/stringparsing_defs.h"
|
||||
#include "simdjson/rvv-vls/numberparsing_defs.h"
|
||||
|
||||
#define SIMDJSON_SKIP_BACKSLASH_SHORT_CIRCUIT 1
|
||||
@@ -0,0 +1,48 @@
|
||||
#ifndef SIMDJSON_RVV_VLS_BITMANIPULATION_H
|
||||
#define SIMDJSON_RVV_VLS_BITMANIPULATION_H
|
||||
|
||||
#ifndef SIMDJSON_CONDITIONAL_INCLUDE
|
||||
#include "simdjson/rvv-vls/base.h"
|
||||
#endif // SIMDJSON_CONDITIONAL_INCLUDE
|
||||
|
||||
namespace simdjson {
|
||||
namespace rvv_vls {
|
||||
namespace {
|
||||
|
||||
// We sometimes call trailing_zero on inputs that are zero,
|
||||
// but the algorithms do not end up using the returned value.
|
||||
// Sadly, sanitizers are not smart enough to figure it out.
|
||||
SIMDJSON_NO_SANITIZE_UNDEFINED
|
||||
// This function can be used safely even if not all bytes have been
|
||||
// initialized.
|
||||
// See issue https://github.com/simdjson/simdjson/issues/1965
|
||||
SIMDJSON_NO_SANITIZE_MEMORY
|
||||
simdjson_inline int trailing_zeroes(uint64_t input_num) {
|
||||
return __builtin_ctzll(input_num);
|
||||
}
|
||||
|
||||
/* result might be undefined when input_num is zero */
|
||||
simdjson_inline uint64_t clear_lowest_bit(uint64_t input_num) {
|
||||
return input_num & (input_num-1);
|
||||
}
|
||||
|
||||
/* result might be undefined when input_num is zero */
|
||||
simdjson_inline int leading_zeroes(uint64_t input_num) {
|
||||
return __builtin_clzll(input_num);
|
||||
}
|
||||
|
||||
simdjson_inline long long int count_ones(uint64_t input_num) {
|
||||
return __builtin_popcountll(input_num);
|
||||
}
|
||||
|
||||
simdjson_inline bool add_overflow(uint64_t value1, uint64_t value2,
|
||||
uint64_t *result) {
|
||||
return __builtin_uaddll_overflow(value1, value2,
|
||||
reinterpret_cast<unsigned long long *>(result));
|
||||
}
|
||||
|
||||
} // unnamed namespace
|
||||
} // namespace rvv_vls
|
||||
} // namespace simdjson
|
||||
|
||||
#endif // SIMDJSON_RVV_VLS_BITMANIPULATION_H
|
||||
@@ -0,0 +1,39 @@
|
||||
#ifndef SIMDJSON_RVV_VLS_BITMASK_H
|
||||
#define SIMDJSON_RVV_VLS_BITMASK_H
|
||||
|
||||
#ifndef SIMDJSON_CONDITIONAL_INCLUDE
|
||||
#include "simdjson/rvv-vls/base.h"
|
||||
#include "simdjson/rvv-vls/intrinsics.h"
|
||||
#endif // SIMDJSON_CONDITIONAL_INCLUDE
|
||||
|
||||
namespace simdjson {
|
||||
namespace rvv_vls {
|
||||
namespace {
|
||||
|
||||
//
|
||||
// Perform a "cumulative bitwise xor," flipping bits each time a 1 is encountered.
|
||||
//
|
||||
// For example, prefix_xor(00100100) == 00011100
|
||||
//
|
||||
simdjson_inline uint64_t prefix_xor(uint64_t bitmask) {
|
||||
#if __riscv_zbc
|
||||
return __riscv_clmul_64(bitmask, ~(uint64_t)0);
|
||||
#elif __riscv_zvbc
|
||||
return __riscv_vmv_x(__riscv_vclmul(__riscv_vmv_s_x_u64m1(bitmask, 1), ~(uint64_t)0, 1));
|
||||
#else
|
||||
bitmask ^= bitmask << 1;
|
||||
bitmask ^= bitmask << 2;
|
||||
bitmask ^= bitmask << 4;
|
||||
bitmask ^= bitmask << 8;
|
||||
bitmask ^= bitmask << 16;
|
||||
bitmask ^= bitmask << 32;
|
||||
#endif
|
||||
return bitmask;
|
||||
}
|
||||
|
||||
} // unnamed namespace
|
||||
} // namespace rvv_vls
|
||||
} // namespace simdjson
|
||||
|
||||
#endif // SIMDJSON_RVV_VLS_BITMASK_H
|
||||
|
||||
@@ -0,0 +1,8 @@
|
||||
#ifndef SIMDJSON_RVV_VLS_BUILDER_H
|
||||
#define SIMDJSON_RVV_VLS_BUILDER_H
|
||||
|
||||
#include "simdjson/rvv-vls/begin.h"
|
||||
#include "simdjson/generic/builder/amalgamated.h"
|
||||
#include "simdjson/rvv-vls/end.h"
|
||||
|
||||
#endif // SIMDJSON_RVV_VLS_BUILDER_H
|
||||
@@ -0,0 +1,5 @@
|
||||
#ifndef SIMDJSON_CONDITIONAL_INCLUDE
|
||||
#include "simdjson/rvv-vls/base.h"
|
||||
#endif // SIMDJSON_CONDITIONAL_INCLUDE
|
||||
|
||||
#undef SIMDJSON_IMPLEMENTATION
|
||||
@@ -0,0 +1,34 @@
|
||||
#ifndef SIMDJSON_RVV_VLS_IMPLEMENTATION_H
|
||||
#define SIMDJSON_RVV_VLS_IMPLEMENTATION_H
|
||||
|
||||
#ifndef SIMDJSON_CONDITIONAL_INCLUDE
|
||||
#include "simdjson/rvv-vls/base.h"
|
||||
#include "simdjson/implementation.h"
|
||||
#endif // SIMDJSON_CONDITIONAL_INCLUDE
|
||||
|
||||
namespace simdjson {
|
||||
namespace rvv_vls {
|
||||
|
||||
/**
|
||||
* @private
|
||||
*/
|
||||
class implementation final : public simdjson::implementation {
|
||||
public:
|
||||
simdjson_inline implementation() : simdjson::implementation(
|
||||
"rvv_vls",
|
||||
"RISC-V V extension",
|
||||
0
|
||||
) {}
|
||||
simdjson_warn_unused error_code create_dom_parser_implementation(
|
||||
size_t capacity,
|
||||
size_t max_length,
|
||||
std::unique_ptr<simdjson::internal::dom_parser_implementation>& dst
|
||||
) const noexcept final;
|
||||
simdjson_warn_unused error_code minify(const uint8_t *buf, size_t len, uint8_t *dst, size_t &dst_len) const noexcept final;
|
||||
simdjson_warn_unused bool validate_utf8(const char *buf, size_t len) const noexcept final;
|
||||
};
|
||||
|
||||
} // namespace rvv_vls
|
||||
} // namespace simdjson
|
||||
|
||||
#endif // SIMDJSON_RVV_VLS_IMPLEMENTATION_H
|
||||
@@ -0,0 +1,32 @@
|
||||
#ifndef SIMDJSON_RVV_VLS_INTRINSICS_H
|
||||
#define SIMDJSON_RVV_VLS_INTRINSICS_H
|
||||
|
||||
#ifndef SIMDJSON_CONDITIONAL_INCLUDE
|
||||
#include "simdjson/rvv-vls/base.h"
|
||||
#endif // SIMDJSON_CONDITIONAL_INCLUDE
|
||||
|
||||
#include <riscv_vector.h>
|
||||
|
||||
#define simdutf_vrgather_u8m1x2(tbl, idx) \
|
||||
__riscv_vcreate_v_u8m1_u8m2( \
|
||||
__riscv_vrgather_vv_u8m1(tbl, __riscv_vget_v_u8m2_u8m1(idx, 0), \
|
||||
__riscv_vsetvlmax_e8m1()), \
|
||||
__riscv_vrgather_vv_u8m1(tbl, __riscv_vget_v_u8m2_u8m1(idx, 1), \
|
||||
__riscv_vsetvlmax_e8m1()))
|
||||
|
||||
#define simdutf_vrgather_u8m1x4(tbl, idx) \
|
||||
__riscv_vcreate_v_u8m1_u8m4( \
|
||||
__riscv_vrgather_vv_u8m1(tbl, __riscv_vget_v_u8m4_u8m1(idx, 0), \
|
||||
__riscv_vsetvlmax_e8m1()), \
|
||||
__riscv_vrgather_vv_u8m1(tbl, __riscv_vget_v_u8m4_u8m1(idx, 1), \
|
||||
__riscv_vsetvlmax_e8m1()), \
|
||||
__riscv_vrgather_vv_u8m1(tbl, __riscv_vget_v_u8m4_u8m1(idx, 2), \
|
||||
__riscv_vsetvlmax_e8m1()), \
|
||||
__riscv_vrgather_vv_u8m1(tbl, __riscv_vget_v_u8m4_u8m1(idx, 3), \
|
||||
__riscv_vsetvlmax_e8m1()))
|
||||
|
||||
#if __riscv_zbc
|
||||
#include <riscv_bitmanip.h>
|
||||
#endif
|
||||
|
||||
#endif // SIMDJSON_RVV_VLS_INTRINSICS_H
|
||||
@@ -0,0 +1,56 @@
|
||||
#ifndef SIMDJSON_RVV_VLS_NUMBERPARSING_DEFS_H
|
||||
#define SIMDJSON_RVV_VLS_NUMBERPARSING_DEFS_H
|
||||
|
||||
#ifndef SIMDJSON_CONDITIONAL_INCLUDE
|
||||
#include "simdjson/rvv-vls/base.h"
|
||||
#include "simdjson/internal/numberparsing_tables.h"
|
||||
#endif // SIMDJSON_CONDITIONAL_INCLUDE
|
||||
|
||||
#include <cstring>
|
||||
|
||||
#ifdef JSON_TEST_NUMBERS // for unit testing
|
||||
void found_invalid_number(const uint8_t *buf);
|
||||
void found_integer(int64_t result, const uint8_t *buf);
|
||||
void found_unsigned_integer(uint64_t result, const uint8_t *buf);
|
||||
void found_float(double result, const uint8_t *buf);
|
||||
#endif
|
||||
|
||||
namespace simdjson {
|
||||
namespace rvv_vls {
|
||||
namespace numberparsing {
|
||||
|
||||
// credit: https://johnnylee-sde.github.io/Fast-numeric-string-to-int/
|
||||
/** @private */
|
||||
static simdjson_inline uint32_t parse_eight_digits_unrolled(const char *chars) {
|
||||
uint64_t val;
|
||||
#if __riscv_misaligned_fast
|
||||
memcpy(&val, chars, sizeof(uint64_t));
|
||||
#else
|
||||
val = __riscv_vmv_x(__riscv_vreinterpret_u64m1(__riscv_vlmul_ext_u8m1(__riscv_vle8_v_u8mf2((uint8_t*)chars, 8))));
|
||||
#endif
|
||||
val = (val & 0x0F0F0F0F0F0F0F0F) * 2561 >> 8;
|
||||
val = (val & 0x00FF00FF00FF00FF) * 6553601 >> 16;
|
||||
return uint32_t((val & 0x0000FFFF0000FFFF) * 42949672960001 >> 32);
|
||||
}
|
||||
|
||||
/** @private */
|
||||
static simdjson_inline uint32_t parse_eight_digits_unrolled(const uint8_t *chars) {
|
||||
return parse_eight_digits_unrolled(reinterpret_cast<const char *>(chars));
|
||||
}
|
||||
|
||||
/** @private */
|
||||
simdjson_inline internal::value128 full_multiplication(uint64_t value1, uint64_t value2) {
|
||||
internal::value128 answer;
|
||||
__uint128_t r = (static_cast<__uint128_t>(value1)) * value2;
|
||||
answer.low = uint64_t(r);
|
||||
answer.high = uint64_t(r >> 64);
|
||||
return answer;
|
||||
}
|
||||
|
||||
} // namespace numberparsing
|
||||
} // namespace rvv_vls
|
||||
} // namespace simdjson
|
||||
|
||||
#define SIMDJSON_SWAR_NUMBER_PARSING 1
|
||||
|
||||
#endif // SIMDJSON_RVV_VLS_NUMBERPARSING_DEFS_H
|
||||
@@ -0,0 +1,8 @@
|
||||
#ifndef SIMDJSON_RVV_VLS_ONDEMAND_H
|
||||
#define SIMDJSON_RVV_VLS_ONDEMAND_H
|
||||
|
||||
#include "simdjson/rvv-vls/begin.h"
|
||||
#include "simdjson/generic/ondemand/amalgamated.h"
|
||||
#include "simdjson/rvv-vls/end.h"
|
||||
|
||||
#endif // SIMDJSON_RVV_VLS_ONDEMAND_H
|
||||
@@ -0,0 +1,370 @@
|
||||
#ifndef SIMDJSON_RVV_VLS_SIMD_H
|
||||
#define SIMDJSON_RVV_VLS_SIMD_H
|
||||
|
||||
#ifndef SIMDJSON_CONDITIONAL_INCLUDE
|
||||
#include "simdjson/rvv-vls/base.h"
|
||||
#include "simdjson/rvv-vls/bitmanipulation.h"
|
||||
#include "simdjson/internal/simdprune_tables.h"
|
||||
#endif // SIMDJSON_CONDITIONAL_INCLUDE
|
||||
|
||||
namespace simdjson {
|
||||
namespace rvv_vls {
|
||||
namespace {
|
||||
namespace simd {
|
||||
|
||||
#if __riscv_v_fixed_vlen >= 512
|
||||
static constexpr size_t VL8 = 512/8;
|
||||
using vint8_t = vint8m1_t __attribute__((riscv_rvv_vector_bits(512)));
|
||||
using vuint8_t = vuint8m1_t __attribute__((riscv_rvv_vector_bits(512)));
|
||||
using vbool_t = vbool8_t __attribute__((riscv_rvv_vector_bits(512/8)));
|
||||
using vbitmask_t = uint64_t;
|
||||
#else
|
||||
static constexpr size_t VL8 = __riscv_v_fixed_vlen/8;
|
||||
using vint8_t = vint8m1_t __attribute__((riscv_rvv_vector_bits(__riscv_v_fixed_vlen)));
|
||||
using vuint8_t = vuint8m1_t __attribute__((riscv_rvv_vector_bits(__riscv_v_fixed_vlen)));
|
||||
using vbool_t = vbool8_t __attribute__((riscv_rvv_vector_bits(__riscv_v_fixed_vlen/8)));
|
||||
#if __riscv_v_fixed_vlen == 128
|
||||
using vbitmask_t = uint16_t;
|
||||
#elif __riscv_v_fixed_vlen == 256
|
||||
using vbitmask_t = uint32_t;
|
||||
#endif
|
||||
#endif
|
||||
|
||||
#if __riscv_v_fixed_vlen == 128
|
||||
using vuint8x64_t = vuint8m4_t __attribute__((riscv_rvv_vector_bits(512)));
|
||||
using vboolx64_t = vbool2_t __attribute__((riscv_rvv_vector_bits(512/8)));
|
||||
#elif __riscv_v_fixed_vlen == 256
|
||||
using vuint8x64_t = vuint8m2_t __attribute__((riscv_rvv_vector_bits(512)));
|
||||
using vboolx64_t = vbool4_t __attribute__((riscv_rvv_vector_bits(512/8)));
|
||||
#else
|
||||
using vuint8x64_t = vuint8m1_t __attribute__((riscv_rvv_vector_bits(512)));
|
||||
using vboolx64_t = vbool8_t __attribute__((riscv_rvv_vector_bits(512/8)));
|
||||
#endif
|
||||
|
||||
template<typename T>
|
||||
struct simd8;
|
||||
|
||||
// SIMD byte mask type (returned by things like eq and gt)
|
||||
template<>
|
||||
struct simd8<bool> {
|
||||
vbool_t value;
|
||||
using bitmask_t = vbitmask_t;
|
||||
static constexpr int SIZE = sizeof(value);
|
||||
|
||||
simdjson_inline simd8(const vbool_t _value) : value(_value) {}
|
||||
simdjson_inline simd8() : simd8(__riscv_vmclr_m_b8(VL8)) {}
|
||||
simdjson_inline simd8(bool _value) : simd8(splat(_value)) {}
|
||||
|
||||
simdjson_inline operator const vbool_t&() const { return value; }
|
||||
simdjson_inline operator vbool_t&() { return value; }
|
||||
|
||||
static simdjson_inline simd8<bool> splat(bool _value) {
|
||||
return __riscv_vreinterpret_b8(__riscv_vmv_v_x_u64m1(((uint64_t)!_value)-1, 1));
|
||||
}
|
||||
|
||||
simdjson_inline vbitmask_t to_bitmask() const {
|
||||
#if __riscv_v_fixed_vlen == 128
|
||||
return __riscv_vmv_x(__riscv_vreinterpret_u16m1(value));
|
||||
#elif __riscv_v_fixed_vlen == 256
|
||||
return __riscv_vmv_x(__riscv_vreinterpret_u32m1(value));
|
||||
#else
|
||||
return __riscv_vmv_x(__riscv_vreinterpret_u64m1(value));
|
||||
#endif
|
||||
}
|
||||
|
||||
// Bit operations
|
||||
simdjson_inline simd8<bool> operator|(const simd8<bool> other) const { return __riscv_vmor(*this, other, VL8); }
|
||||
simdjson_inline simd8<bool> operator&(const simd8<bool> other) const { return __riscv_vmand(*this, other, VL8); }
|
||||
simdjson_inline simd8<bool> operator^(const simd8<bool> other) const { return __riscv_vmxor(*this, other, VL8); }
|
||||
simdjson_inline simd8<bool> bit_andnot(const simd8<bool> other) const { return __riscv_vmandn(other, *this, VL8); }
|
||||
simdjson_inline simd8<bool> operator~() const { return __riscv_vmnot(*this, VL8); }
|
||||
simdjson_inline simd8<bool>& operator|=(const simd8<bool> other) { auto this_cast = static_cast<simd8<bool>*>(this); *this_cast = *this_cast | other; return *this_cast; }
|
||||
simdjson_inline simd8<bool>& operator&=(const simd8<bool> other) { auto this_cast = static_cast<simd8<bool>*>(this); *this_cast = *this_cast & other; return *this_cast; }
|
||||
simdjson_inline simd8<bool>& operator^=(const simd8<bool> other) { auto this_cast = static_cast<simd8<bool>*>(this); *this_cast = *this_cast ^ other; return *this_cast; }
|
||||
};
|
||||
|
||||
// Unsigned bytes
|
||||
template<>
|
||||
struct simd8<uint8_t> {
|
||||
|
||||
vuint8_t value;
|
||||
static constexpr int SIZE = sizeof(value);
|
||||
|
||||
simdjson_inline simd8(const vuint8_t _value) : value(_value) {}
|
||||
simdjson_inline simd8() : simd8(zero()) {}
|
||||
simdjson_inline simd8(const uint8_t values[VL8]) : simd8(load(values)) {}
|
||||
simdjson_inline simd8(uint8_t _value) : simd8(splat(_value)) {}
|
||||
simdjson_inline simd8(simd8<bool> mask) : value(__riscv_vmerge_vxm_u8m1(zero(), -1, (vbool_t)mask, VL8)) {}
|
||||
|
||||
simdjson_inline operator const vuint8_t&() const { return this->value; }
|
||||
simdjson_inline operator vuint8_t&() { return this->value; }
|
||||
|
||||
simdjson_inline simd8(
|
||||
uint8_t v0, uint8_t v1, uint8_t v2, uint8_t v3, uint8_t v4, uint8_t v5, uint8_t v6, uint8_t v7,
|
||||
uint8_t v8, uint8_t v9, uint8_t v10, uint8_t v11, uint8_t v12, uint8_t v13, uint8_t v14, uint8_t v15
|
||||
) : simd8(vuint8_t{
|
||||
v0, v1, v2, v3, v4, v5, v6, v7,
|
||||
v8, v9, v10,v11,v12,v13,v14,v15
|
||||
}) {}
|
||||
|
||||
// Repeat 16 values as many times as necessary (usually for lookup tables)
|
||||
simdjson_inline static simd8<uint8_t> repeat_16(
|
||||
uint8_t v0, uint8_t v1, uint8_t v2, uint8_t v3, uint8_t v4, uint8_t v5, uint8_t v6, uint8_t v7,
|
||||
uint8_t v8, uint8_t v9, uint8_t v10, uint8_t v11, uint8_t v12, uint8_t v13, uint8_t v14, uint8_t v15
|
||||
) {
|
||||
return simd8<uint8_t>(
|
||||
v0, v1, v2, v3, v4, v5, v6, v7,
|
||||
v8, v9, v10,v11,v12,v13,v14,v15
|
||||
);
|
||||
}
|
||||
|
||||
static simdjson_inline vuint8_t splat(uint8_t _value) { return __riscv_vmv_v_x_u8m1(_value, VL8); }
|
||||
static simdjson_inline vuint8_t zero() { return splat(0); }
|
||||
static simdjson_inline vuint8_t load(const uint8_t values[VL8]) { return __riscv_vle8_v_u8m1(values, VL8); }
|
||||
|
||||
// Bit operations
|
||||
simdjson_inline simd8<uint8_t> operator|(const simd8<uint8_t> other) const { return __riscv_vor_vv_u8m1( value, other, VL8); }
|
||||
simdjson_inline simd8<uint8_t> operator&(const simd8<uint8_t> other) const { return __riscv_vand_vv_u8m1( value, other, VL8); }
|
||||
simdjson_inline simd8<uint8_t> operator^(const simd8<uint8_t> other) const { return __riscv_vxor_vv_u8m1( value, other, VL8); }
|
||||
simdjson_inline simd8<uint8_t> operator~() const { return __riscv_vnot_v_u8m1(value, VL8); }
|
||||
#if __riscv_zvbb
|
||||
simdjson_inline simd8<uint8_t> bit_andnot(const simd8<uint8_t> other) const { return __riscv_vandn_vv_u8m1(other, value, VL8); }
|
||||
#else
|
||||
simdjson_inline simd8<uint8_t> bit_andnot(const simd8<uint8_t> other) const { return other & ~*this; }
|
||||
#endif
|
||||
simdjson_inline simd8<uint8_t>& operator|=(const simd8<uint8_t> other) { value = *this | other; return *this; }
|
||||
simdjson_inline simd8<uint8_t>& operator&=(const simd8<uint8_t> other) { value = *this & other; return *this; }
|
||||
simdjson_inline simd8<uint8_t>& operator^=(const simd8<uint8_t> other) { value = *this ^ other; return *this; }
|
||||
|
||||
simdjson_inline simd8<bool> operator==(const simd8<uint8_t> other) const { return __riscv_vmseq(value, other, VL8); }
|
||||
simdjson_inline simd8<bool> operator==(uint8_t other) const { return __riscv_vmseq(value, other, VL8); }
|
||||
|
||||
template<int N=1>
|
||||
simdjson_inline simd8<uint8_t> prev(const simd8<uint8_t> prev_chunk) const {
|
||||
return __riscv_vslideup(__riscv_vslidedown(prev_chunk, VL8-N, VL8), value, N, VL8);
|
||||
}
|
||||
|
||||
// Store to array
|
||||
simdjson_inline void store(uint8_t dst[VL8]) const { return __riscv_vse8(dst, value, VL8); }
|
||||
|
||||
// Saturated math
|
||||
simdjson_inline simd8<uint8_t> saturating_add(const simd8<uint8_t> other) const { return __riscv_vsaddu(value, other, VL8); }
|
||||
simdjson_inline simd8<uint8_t> saturating_sub(const simd8<uint8_t> other) const { return __riscv_vssubu(value, other, VL8); }
|
||||
|
||||
// Addition/subtraction are the same for signed and unsigned
|
||||
simdjson_inline simd8<uint8_t> operator+(const simd8<uint8_t> other) const { return __riscv_vadd(value, other, VL8); }
|
||||
simdjson_inline simd8<uint8_t> operator-(const simd8<uint8_t> other) const { return __riscv_vsub(value, other, VL8); }
|
||||
simdjson_inline simd8<uint8_t>& operator+=(const simd8<uint8_t> other) { value = *this + other; return *this; }
|
||||
simdjson_inline simd8<uint8_t>& operator-=(const simd8<uint8_t> other) { value = *this - other; return *this; }
|
||||
|
||||
// Order-specific operations
|
||||
simdjson_inline simd8<bool> operator<=(const simd8<uint8_t> other) const { return __riscv_vmsleu(value, other, VL8); }
|
||||
simdjson_inline simd8<bool> operator>=(const simd8<uint8_t> other) const { return __riscv_vmsgeu(value, other, VL8); }
|
||||
simdjson_inline simd8<bool> operator<(const simd8<uint8_t> other) const { return __riscv_vmsltu(value, other, VL8); }
|
||||
simdjson_inline simd8<bool> operator>(const simd8<uint8_t> other) const { return __riscv_vmsgtu(value, other, VL8); }
|
||||
|
||||
// Same as >, but instead of guaranteeing all 1's == true, false = 0 and true = nonzero.
|
||||
simdjson_inline simd8<uint8_t> gt_bits(const simd8<uint8_t> other) const { return simd8<uint8_t>(*this > other); }
|
||||
// Same as <, but instead of guaranteeing all 1's == true, false = 0 and true = nonzero.
|
||||
simdjson_inline simd8<uint8_t> lt_bits(const simd8<uint8_t> other) const { return simd8<uint8_t>(*this < other); }
|
||||
|
||||
// Bit-specific operations
|
||||
simdjson_inline bool any_bits_set_anywhere() const {
|
||||
return __riscv_vfirst(__riscv_vmsne(value, 0, VL8), VL8) >= 0;
|
||||
}
|
||||
simdjson_inline bool any_bits_set_anywhere(simd8<uint8_t> bits) const { return (*this & bits).any_bits_set_anywhere(); }
|
||||
template<int N>
|
||||
simdjson_inline simd8<uint8_t> shr() const { return __riscv_vsrl(value, N, VL8); }
|
||||
template<int N>
|
||||
simdjson_inline simd8<uint8_t> shl() const { return __riscv_vsll(value, N, VL8); }
|
||||
|
||||
|
||||
// Perform a lookup assuming the value is between 0 and 16 (undefined behavior for out of range values)
|
||||
template<typename L>
|
||||
simdjson_inline simd8<L> lookup_16(simd8<L> lookup_table) const {
|
||||
return __riscv_vrgather(lookup_table, value, VL8);
|
||||
}
|
||||
|
||||
// compress inactive elements, to match AVX-512 behavior
|
||||
template<typename L>
|
||||
simdjson_inline void compress(vbitmask_t mask, L * output) const {
|
||||
mask = (vbitmask_t)~mask;
|
||||
#if __riscv_v_fixed_vlen == 128
|
||||
vbool8_t m = __riscv_vreinterpret_b8(__riscv_vmv_s_x_u16m1(mask, 1));
|
||||
#elif __riscv_v_fixed_vlen == 256
|
||||
vbool8_t m = __riscv_vreinterpret_b8(__riscv_vmv_s_x_u32m1(mask, 1));
|
||||
#else
|
||||
vbool8_t m = __riscv_vreinterpret_b8(__riscv_vmv_s_x_u64m1(mask, 1));
|
||||
#endif
|
||||
__riscv_vse8_v_u8m1(output, __riscv_vcompress(value, m, VL8), count_ones(mask));
|
||||
}
|
||||
|
||||
template<typename L>
|
||||
simdjson_inline simd8<L> lookup_16(
|
||||
L replace0, L replace1, L replace2, L replace3,
|
||||
L replace4, L replace5, L replace6, L replace7,
|
||||
L replace8, L replace9, L replace10, L replace11,
|
||||
L replace12, L replace13, L replace14, L replace15) const {
|
||||
return lookup_16(simd8<L>::repeat_16(
|
||||
replace0, replace1, replace2, replace3,
|
||||
replace4, replace5, replace6, replace7,
|
||||
replace8, replace9, replace10, replace11,
|
||||
replace12, replace13, replace14, replace15
|
||||
));
|
||||
}
|
||||
};
|
||||
|
||||
// Signed bytes
|
||||
template<>
|
||||
struct simd8<int8_t> {
|
||||
vint8_t value;
|
||||
static constexpr int SIZE = sizeof(value);
|
||||
|
||||
simdjson_inline simd8(const vint8_t _value) : value(_value) {}
|
||||
simdjson_inline simd8() : simd8(zero()) {}
|
||||
simdjson_inline simd8(const int8_t values[VL8]) : simd8(load(values)) {}
|
||||
simdjson_inline simd8(int8_t _value) : simd8(splat(_value)) {}
|
||||
|
||||
simdjson_inline operator const vint8_t&() const { return this->value; }
|
||||
simdjson_inline operator vint8_t&() { return this->value; }
|
||||
|
||||
simdjson_inline simd8(
|
||||
int8_t v0, int8_t v1, int8_t v2, int8_t v3, int8_t v4, int8_t v5, int8_t v6, int8_t v7,
|
||||
int8_t v8, int8_t v9, int8_t v10, int8_t v11, int8_t v12, int8_t v13, int8_t v14, int8_t v15
|
||||
) : simd8(vint8_t{
|
||||
v0, v1, v2, v3, v4, v5, v6, v7,
|
||||
v8, v9, v10,v11,v12,v13,v14,v15
|
||||
}) {}
|
||||
|
||||
// Repeat 16 values as many times as necessary (usually for lookup tables)
|
||||
simdjson_inline static simd8<int8_t> repeat_16(
|
||||
int8_t v0, int8_t v1, int8_t v2, int8_t v3, int8_t v4, int8_t v5, int8_t v6, int8_t v7,
|
||||
int8_t v8, int8_t v9, int8_t v10, int8_t v11, int8_t v12, int8_t v13, int8_t v14, int8_t v15
|
||||
) {
|
||||
return simd8<int8_t>(
|
||||
v0, v1, v2, v3, v4, v5, v6, v7,
|
||||
v8, v9, v10,v11,v12,v13,v14,v15
|
||||
);
|
||||
}
|
||||
|
||||
static simdjson_inline vint8_t splat(int8_t _value) { return __riscv_vmv_v_x_i8m1(_value, VL8); }
|
||||
static simdjson_inline vint8_t zero() { return splat(0); }
|
||||
static simdjson_inline vint8_t load(const int8_t values[VL8]) { return __riscv_vle8_v_i8m1(values, VL8); }
|
||||
|
||||
|
||||
simdjson_inline void store(int8_t dst[VL8]) const { return __riscv_vse8(dst, value, VL8); }
|
||||
|
||||
// Explicit conversion to/from unsigned
|
||||
simdjson_inline explicit simd8(const vuint8_t other): simd8(__riscv_vreinterpret_i8m1(other)) {}
|
||||
simdjson_inline explicit operator simd8<uint8_t>() const { return __riscv_vreinterpret_u8m1(value); }
|
||||
|
||||
// Math
|
||||
simdjson_inline simd8<int8_t> operator+(const simd8<int8_t> other) const { return __riscv_vadd(value, other, VL8); }
|
||||
simdjson_inline simd8<int8_t> operator-(const simd8<int8_t> other) const { return __riscv_vsub(value, other, VL8); }
|
||||
simdjson_inline simd8<int8_t>& operator+=(const simd8<int8_t> other) { value = *this + other; return *this; }
|
||||
simdjson_inline simd8<int8_t>& operator-=(const simd8<int8_t> other) { value = *this - other; return *this; }
|
||||
|
||||
// Order-sensitive comparisons
|
||||
simdjson_inline simd8<int8_t> max_val( const simd8<int8_t> other) const { return __riscv_vmax( value, other, VL8); }
|
||||
simdjson_inline simd8<int8_t> min_val( const simd8<int8_t> other) const { return __riscv_vmin( value, other, VL8); }
|
||||
simdjson_inline simd8<bool> operator>( const simd8<int8_t> other) const { return __riscv_vmsgt(value, other, VL8); }
|
||||
simdjson_inline simd8<bool> operator<( const simd8<int8_t> other) const { return __riscv_vmslt(value, other, VL8); }
|
||||
simdjson_inline simd8<bool> operator==(const simd8<int8_t> other) const { return __riscv_vmseq(value, other, VL8); }
|
||||
|
||||
template<int N=1>
|
||||
simdjson_inline simd8<int8_t> prev(const simd8<int8_t> prev_chunk) const {
|
||||
return __riscv_vslideup(__riscv_vslidedown(prev_chunk, VL8-N, VL8), value, N, VL8);
|
||||
}
|
||||
|
||||
// Perform a lookup assuming no value is larger than 16
|
||||
template<typename L>
|
||||
simdjson_inline simd8<L> lookup_16(simd8<L> lookup_table) const {
|
||||
return __riscv_vrgather(lookup_table, value, VL8);
|
||||
}
|
||||
template<typename L>
|
||||
simdjson_inline simd8<L> lookup_16(
|
||||
L replace0, L replace1, L replace2, L replace3,
|
||||
L replace4, L replace5, L replace6, L replace7,
|
||||
L replace8, L replace9, L replace10, L replace11,
|
||||
L replace12, L replace13, L replace14, L replace15) const {
|
||||
return lookup_16(simd8<L>::repeat_16(
|
||||
replace0, replace1, replace2, replace3,
|
||||
replace4, replace5, replace6, replace7,
|
||||
replace8, replace9, replace10, replace11,
|
||||
replace12, replace13, replace14, replace15
|
||||
));
|
||||
}
|
||||
};
|
||||
|
||||
template<typename T>
|
||||
struct simd8x64;
|
||||
template<>
|
||||
struct simd8x64<uint8_t> {
|
||||
static constexpr int NUM_CHUNKS = 64 / sizeof(simd8<uint8_t>);
|
||||
vuint8x64_t value;
|
||||
|
||||
#if __riscv_v_fixed_vlen >= 512
|
||||
template<int idx> simd8<uint8_t> get() const { return value; }
|
||||
#else
|
||||
template<int idx> simd8<uint8_t> get() const { return __riscv_vget_u8m1(value, idx); }
|
||||
#endif
|
||||
|
||||
simdjson_inline operator const vuint8x64_t&() const { return this->value; }
|
||||
simdjson_inline operator vuint8x64_t&() { return this->value; }
|
||||
|
||||
simd8x64(const simd8x64<uint8_t>& o) = delete; // no copy allowed
|
||||
simd8x64<uint8_t>& operator=(const simd8<uint8_t>& other) = delete; // no assignment allowed
|
||||
simd8x64() = delete; // no default constructor allowed
|
||||
|
||||
#if __riscv_v_fixed_vlen == 128
|
||||
simdjson_inline simd8x64(const uint8_t *ptr, size_t n = 64) : value(__riscv_vle8_v_u8m4(ptr, n)) {}
|
||||
#elif __riscv_v_fixed_vlen == 256
|
||||
simdjson_inline simd8x64(const uint8_t *ptr, size_t n = 64) : value(__riscv_vle8_v_u8m2(ptr, n)) {}
|
||||
#else
|
||||
simdjson_inline simd8x64(const uint8_t *ptr, size_t n = 64) : value(__riscv_vle8_v_u8m1(ptr, n)) {}
|
||||
#endif
|
||||
|
||||
simdjson_inline void store(uint8_t ptr[64]) const {
|
||||
__riscv_vse8(ptr, value, 64);
|
||||
}
|
||||
|
||||
simdjson_inline bool is_ascii() const {
|
||||
#if __riscv_v_fixed_vlen == 128
|
||||
return __riscv_vfirst(__riscv_vmslt(__riscv_vreinterpret_i8m4(value), 0, 64), 64) < 0;
|
||||
#elif __riscv_v_fixed_vlen == 256
|
||||
return __riscv_vfirst(__riscv_vmslt(__riscv_vreinterpret_i8m2(value), 0, 64), 64) < 0;
|
||||
#else
|
||||
return __riscv_vfirst(__riscv_vmslt(__riscv_vreinterpret_i8m1(value), 0, 64), 64) < 0;
|
||||
#endif
|
||||
}
|
||||
|
||||
// compress inactive elements, to match AVX-512 behavior
|
||||
simdjson_inline uint64_t compress(uint64_t mask, uint8_t * output) const {
|
||||
mask = ~mask;
|
||||
#if __riscv_v_fixed_vlen == 128
|
||||
vboolx64_t m = __riscv_vreinterpret_b2(__riscv_vmv_s_x_u64m1(mask, 1));
|
||||
#elif __riscv_v_fixed_vlen == 256
|
||||
vboolx64_t m = __riscv_vreinterpret_b4(__riscv_vmv_s_x_u64m1(mask, 1));
|
||||
#else
|
||||
vboolx64_t m = __riscv_vreinterpret_b8(__riscv_vmv_s_x_u64m1(mask, 1));
|
||||
#endif
|
||||
size_t cnt = count_ones(mask);
|
||||
__riscv_vse8(output, __riscv_vcompress(value, m, 64), cnt);
|
||||
return cnt;
|
||||
}
|
||||
|
||||
simdjson_inline uint64_t eq(const uint8_t m) const {
|
||||
return __riscv_vmv_x(__riscv_vreinterpret_u64m1(__riscv_vmseq(value, m, 64)));
|
||||
}
|
||||
|
||||
simdjson_inline uint64_t lteq(const uint8_t m) const {
|
||||
return __riscv_vmv_x(__riscv_vreinterpret_u64m1(__riscv_vmsleu(value, m, 64)));
|
||||
}
|
||||
}; // struct simd8x64<uint8_t>
|
||||
|
||||
} // namespace simd
|
||||
} // unnamed namespace
|
||||
} // namespace rvv_vls
|
||||
} // namespace simdjson
|
||||
|
||||
#endif // SIMDJSON_RVV_VLS_SIMD_H
|
||||
@@ -0,0 +1,58 @@
|
||||
#ifndef SIMDJSON_RVV_VLS_STRINGPARSING_DEFS_H
|
||||
#define SIMDJSON_RVV_VLS_STRINGPARSING_DEFS_H
|
||||
|
||||
#ifndef SIMDJSON_CONDITIONAL_INCLUDE
|
||||
#include "simdjson/rvv-vls/base.h"
|
||||
#endif // SIMDJSON_CONDITIONAL_INCLUDE
|
||||
|
||||
namespace simdjson {
|
||||
namespace rvv_vls {
|
||||
namespace {
|
||||
|
||||
using namespace simd;
|
||||
|
||||
// Holds backslashes and quotes locations.
|
||||
struct backslash_and_quote {
|
||||
public:
|
||||
static constexpr uint64_t BYTES_PROCESSED = sizeof(simd8<uint8_t>);
|
||||
simdjson_inline backslash_and_quote copy_and_find(const uint8_t *src, uint8_t *dst);
|
||||
|
||||
simdjson_inline bool has_quote_first() { return ((bs_bits - 1) & quote_bits) != 0; }
|
||||
simdjson_inline bool has_backslash() { return ((quote_bits - 1) & bs_bits) != 0; }
|
||||
simdjson_inline int quote_index() { return trailing_zeroes(quote_bits); }
|
||||
simdjson_inline int backslash_index() { return trailing_zeroes(bs_bits); }
|
||||
|
||||
uint64_t bs_bits;
|
||||
uint64_t quote_bits;
|
||||
}; // struct backslash_and_quote
|
||||
|
||||
simdjson_inline backslash_and_quote backslash_and_quote::copy_and_find(const uint8_t *src, uint8_t *dst) {
|
||||
static_assert(SIMDJSON_PADDING >= (BYTES_PROCESSED - 1), "backslash and quote finder must process fewer than SIMDJSON_PADDING bytes");
|
||||
simd8<uint8_t> v(src);
|
||||
v.store(dst);
|
||||
return { (v == '\\').to_bitmask(), (v == '"').to_bitmask() };
|
||||
}
|
||||
|
||||
struct escaping {
|
||||
static constexpr uint64_t BYTES_PROCESSED = sizeof(simd8<uint8_t>);
|
||||
simdjson_inline static escaping copy_and_find(const uint8_t *src, uint8_t *dst);
|
||||
|
||||
simdjson_inline bool has_escape() { return escape_bits != 0; }
|
||||
simdjson_inline int escape_index() { return trailing_zeroes(escape_bits) / 4; }
|
||||
|
||||
uint64_t escape_bits;
|
||||
}; // struct escaping
|
||||
|
||||
simdjson_inline escaping escaping::copy_and_find(const uint8_t *src, uint8_t *dst) {
|
||||
static_assert(SIMDJSON_PADDING >= (BYTES_PROCESSED - 1), "escaping finder must process fewer than SIMDJSON_PADDING bytes");
|
||||
simd8<uint8_t> v(src);
|
||||
v.store(dst);
|
||||
return { ((v == '"') | (v == '\\') | (v == 32)).to_bitmask() };
|
||||
}
|
||||
|
||||
|
||||
} // unnamed namespace
|
||||
} // namespace rvv_vls
|
||||
} // namespace simdjson
|
||||
|
||||
#endif // SIMDJSON_RVV_VLS_STRINGPARSING_DEFS_H
|
||||
@@ -4,7 +4,7 @@
|
||||
#define SIMDJSON_SIMDJSON_VERSION_H
|
||||
|
||||
/** The version of simdjson being used (major.minor.revision) */
|
||||
#define SIMDJSON_VERSION "4.2.4"
|
||||
#define SIMDJSON_VERSION "4.3.0"
|
||||
|
||||
namespace simdjson {
|
||||
enum {
|
||||
@@ -15,11 +15,11 @@ enum {
|
||||
/**
|
||||
* The minor version (major.MINOR.revision) of simdjson being used.
|
||||
*/
|
||||
SIMDJSON_VERSION_MINOR = 2,
|
||||
SIMDJSON_VERSION_MINOR = 3,
|
||||
/**
|
||||
* The revision (major.minor.REVISION) of simdjson being used.
|
||||
*/
|
||||
SIMDJSON_VERSION_REVISION = 4
|
||||
SIMDJSON_VERSION_REVISION = 0
|
||||
};
|
||||
} // namespace simdjson
|
||||
|
||||
|
||||
@@ -265,6 +265,7 @@ namespace simd {
|
||||
static constexpr int NUM_CHUNKS = 64 / sizeof(simd8<T>);
|
||||
static_assert(NUM_CHUNKS == 4, "Westmere kernel should use four registers per 64-byte block.");
|
||||
const simd8<T> chunks[NUM_CHUNKS];
|
||||
template<int idx> simd8<uint8_t> get() const { return idx < NUM_CHUNKS ? chunks[idx] : simd8<T>(); }
|
||||
|
||||
simd8x64(const simd8x64<T>& o) = delete; // no copy allowed
|
||||
simd8x64<T>& operator=(const simd8<T>& other) = delete; // no assignment allowed
|
||||
|
||||
+11
-4
@@ -54,15 +54,15 @@ Importantly, we build the experimental LLVM compiler based on the current state
|
||||
```bash
|
||||
CXX=clang++ cmake -B buildreflect -D SIMDJSON_STATIC_REFLECTION=ON -DSIMDJSON_DEVELOPER_MODE=ON
|
||||
```
|
||||
This only needs to be done once.
|
||||
This only needs to be done once. To build the Rust code, add `-D SIMDJSON_USE_RUST=ON`. Note that you should have Rust on your system as a prerequisite for this option to be meaningful.
|
||||
|
||||
5. Build the code...
|
||||
```bash
|
||||
cmake --build buildreflect --target benchmark_serialization_citm_catalog benchmark_serialization_twitter
|
||||
cmake --build buildreflect --target benchmark_serialization_citm_catalog benchmark_serialization_twitter benchmark_parsing_twitter benchmark_parsing_citm
|
||||
```
|
||||
This is sufficient if you only mean to run the benchmarks (and skip the tests).
|
||||
|
||||
|
||||
6. Run the tests...
|
||||
6. Run the tests... (optional)
|
||||
```bash
|
||||
cmake --build buildreflect
|
||||
ctest --test-dir buildreflect --output-on-failure
|
||||
@@ -71,9 +71,16 @@ ctest --test-dir buildreflect --output-on-failure
|
||||
7. Run the benchmarks.
|
||||
```bash
|
||||
./buildreflect/benchmark/static_reflect/citm_catalog_benchmark/benchmark_serialization_citm_catalog
|
||||
./buildreflect/benchmark/static_reflect/citm_catalog_benchmark/benchmark_parsing_citm
|
||||
./buildreflect/benchmark/static_reflect/twitter_benchmark/benchmark_serialization_twitter
|
||||
./buildreflect/benchmark/static_reflect/twitter_benchmark/benchmark_parsing_twitter
|
||||
```
|
||||
|
||||
These benchmarks should print performance counters *if* run in privileged mode (e.g., run under sudo).
|
||||
If you run the code inside a docker container, under Linux, it should have access to the performance
|
||||
counters if they are enabled on the host. Note that they are usually disabled in a cloud setting
|
||||
unless you have so-called metal access.
|
||||
|
||||
You can modify the source code with your favorite editor and run again steps 5 (Build the code) and 6 (Run the tests) and 7 (Run the benchmark). Importantly, you should remain in the docker shell.
|
||||
|
||||
You can create a new docker shell at any time by running step 3 (bash script).
|
||||
|
||||
+27
-14
@@ -13,6 +13,18 @@ import datetime
|
||||
import json
|
||||
from typing import Dict, List, Optional, Set, TextIO, Union, cast
|
||||
|
||||
# Pre-compile regex patterns for performance
|
||||
pragma_once_re = re.compile(r'^#pragma once$')
|
||||
ifndef_conditional_re = re.compile(r'^#ifndef\s+SIMDJSON_CONDITIONAL_INCLUDE\s*$')
|
||||
endif_conditional_re = re.compile(r'^#endif\s*//\s*SIMDJSON_CONDITIONAL_INCLUDE\s*$')
|
||||
include_re = re.compile(r'^#include\s+["<]([^">]*)[">]')
|
||||
define_implementation_re = re.compile(r'^#define\s+SIMDJSON_IMPLEMENTATION\s+(.+)$')
|
||||
undef_implementation_re = re.compile(r'^#undef\s+SIMDJSON_IMPLEMENTATION\s*$')
|
||||
simdjson_implementation_re = re.compile(r'\bSIMDJSON_IMPLEMENTATION\b')
|
||||
define_conditional_re = re.compile(r'^#define\s+SIMDJSON_CONDITIONAL_INCLUDE\s*$')
|
||||
undef_conditional_re = re.compile(r'^#undef\s+SIMDJSON_CONDITIONAL_INCLUDE\s*$')
|
||||
version_re = re.compile(r'\d+\.\d+\.\d+')
|
||||
|
||||
# Check for Python 3, this does not actually work.
|
||||
if sys.version_info < (3, 0):
|
||||
sys.stdout.write("Sorry, requires Python 3.x or better\n")
|
||||
@@ -108,7 +120,8 @@ else:
|
||||
RelativeRoot = str # Literal['src','include'] # Literal not supported in Python 3.7 (CI)
|
||||
RELATIVE_ROOTS: List[RelativeRoot] = ['src', 'include' ]
|
||||
Implementation = str # Literal['arm64', 'fallback', 'haswell', 'icelake', 'ppc64', 'westmere', 'lsx', 'lasx'] # Literal not supported in Python 3.7 (CI)
|
||||
IMPLEMENTATIONS: List[Implementation] = [ 'arm64', 'haswell', 'icelake', 'lasx', 'lsx', 'ppc64', 'westmere', 'fallback' ]
|
||||
IMPLEMENTATIONS: List[Implementation] = [ 'arm64', 'haswell', 'icelake', 'lasx', 'lsx', 'ppc64', 'rvv-vls', 'westmere', 'fallback' ]
|
||||
implementation_re = re.compile(f'(^|/)({"|".join(IMPLEMENTATIONS)})')
|
||||
GENERIC_INCLUDE = "simdjson/generic"
|
||||
GENERIC_SRC = "generic"
|
||||
BUILTIN = "simdjson/builtin"
|
||||
@@ -199,7 +212,7 @@ class SimdjsonFile:
|
||||
|
||||
@property
|
||||
def implementation(self) -> Optional[Implementation]:
|
||||
match = re.search(f'(^|/)({"|".join(IMPLEMENTATIONS)})', self.include_path)
|
||||
match = implementation_re.search(self.include_path)
|
||||
if match:
|
||||
return cast(Implementation, str(match.group(2)))
|
||||
|
||||
@@ -417,23 +430,23 @@ class Amalgamator:
|
||||
line = line.rstrip('\n')
|
||||
|
||||
# Ignore #pragma once, it causes warnings if it ends up in a .cpp file
|
||||
if re.search(r'^#pragma once$', line):
|
||||
if pragma_once_re.search(line):
|
||||
continue
|
||||
|
||||
# Ignore lines inside #ifndef SIMDJSON_CONDITIONAL_INCLUDE
|
||||
if re.search(r'^#ifndef\s+SIMDJSON_CONDITIONAL_INCLUDE\s*$', line):
|
||||
if ifndef_conditional_re.search(line):
|
||||
assert file.is_conditional_include, f"Error: File '{file}' uses '#ifndef SIMDJSON_CONDITIONAL_INCLUDE', but it's not an amalgamated file. Conditional includes are only for amalgamated files. {rules}"
|
||||
assert self.in_conditional_include_block, f"Error: File '{file}' uses '#ifndef SIMDJSON_CONDITIONAL_INCLUDE' without a prior '#define SIMDJSON_CONDITIONAL_INCLUDE'. Ensure the define comes first. Stack: {self.include_stack}. {rules}"
|
||||
assert not self.editor_only_region, f"Error: File '{file}' uses '#ifndef SIMDJSON_CONDITIONAL_INCLUDE' twice in a row. Ensure conditional blocks are properly nested and closed. {rules}"
|
||||
self.editor_only_region = True
|
||||
|
||||
# Handle ignored lines (and ending ignore blocks)
|
||||
end_ignore = re.search(r'^#endif\s*//\s*SIMDJSON_CONDITIONAL_INCLUDE\s*$', line)
|
||||
end_ignore = endif_conditional_re.search(line)
|
||||
if self.editor_only_region:
|
||||
self.write(f"/* amalgamation skipped (editor-only): {line} */")
|
||||
|
||||
# Add the editor-only include so we can check dependencies.h for completeness later
|
||||
included = re.search(r'^#include\s+["<]([^">]*)[">]', line)
|
||||
included = include_re.search(line)
|
||||
if included:
|
||||
included_file = self.repository[included.group(1)]
|
||||
if included_file:
|
||||
@@ -445,7 +458,7 @@ class Amalgamator:
|
||||
assert not end_ignore, f"Error: File '{file}' has '#endif // SIMDJSON_CONDITIONAL_INCLUDE' without a matching '#ifndef'. Ensure proper conditional block structure. {rules}"
|
||||
|
||||
# Handle #include lines
|
||||
included = re.search(r'^#include\s+["<]([^">]*)[">]', line)
|
||||
included = include_re.search(line)
|
||||
if included:
|
||||
# we explicitly include simdjson headers, one time each (unless they are generic, in which case multiple times is fine)
|
||||
included_file = self.repository[included.group(1)]
|
||||
@@ -455,7 +468,7 @@ class Amalgamator:
|
||||
continue
|
||||
|
||||
# Handle defining and replacing SIMDJSON_IMPLEMENTATION
|
||||
defined = re.search(r'^#define\s+SIMDJSON_IMPLEMENTATION\s+(.+)$', line)
|
||||
defined = define_implementation_re.search(line)
|
||||
if defined:
|
||||
old_implementation = self.implementation
|
||||
self.implementation = defined.group(1)
|
||||
@@ -463,24 +476,24 @@ class Amalgamator:
|
||||
self.write(f'/* defining SIMDJSON_IMPLEMENTATION to "{self.implementation}" */')
|
||||
else:
|
||||
self.write(f'/* redefining SIMDJSON_IMPLEMENTATION from "{old_implementation}" to "{self.implementation}" */')
|
||||
elif re.search(r'^#undef\s+SIMDJSON_IMPLEMENTATION\s*$', line):
|
||||
elif undef_implementation_re.search(line):
|
||||
# Don't include #undef SIMDJSON_IMPLEMENTATION since we're handling it ourselves
|
||||
self.write(f'/* undefining SIMDJSON_IMPLEMENTATION from "{self.implementation}" */')
|
||||
self.implementation = None
|
||||
elif re.search(r'\bSIMDJSON_IMPLEMENTATION\b', line) and file.include_path != IMPLEMENTATION_DETECTION_H:
|
||||
elif self.implementation and file.include_path != IMPLEMENTATION_DETECTION_H:
|
||||
# copy the line, with SIMDJSON_IMPLEMENTATION replace to what it is currently defined to
|
||||
assert self.implementation, f"Error: In '{file}', line '{line}' uses SIMDJSON_IMPLEMENTATION, but it's not defined. Ensure SIMDJSON_IMPLEMENTATION is set before use. {rules}"
|
||||
line = re.sub(r'\bSIMDJSON_IMPLEMENTATION\b',self.implementation,line)
|
||||
line = simdjson_implementation_re.sub(self.implementation, line)
|
||||
|
||||
# Handle defining and undefining SIMDJSON_CONDITIONAL_INCLUDE
|
||||
defined = re.search(r'^#define\s+SIMDJSON_CONDITIONAL_INCLUDE\s*$', line)
|
||||
defined = define_conditional_re.search(line)
|
||||
if defined:
|
||||
assert not file.is_conditional_include, f"Error: Amalgamated file '{file}' defines SIMDJSON_CONDITIONAL_INCLUDE, which is not allowed. Only non-amalgamated files can define it. {rules}"
|
||||
assert not self.in_conditional_include_block, f"Error: File '{file}' redefines SIMDJSON_CONDITIONAL_INCLUDE while already in a conditional block. Avoid redefinition. {rules}"
|
||||
self.in_conditional_include_block = True
|
||||
self.found_includes_per_conditional_block.clear()
|
||||
self.write(f'/* defining SIMDJSON_CONDITIONAL_INCLUDE */')
|
||||
elif re.search(r'^#undef\s+SIMDJSON_CONDITIONAL_INCLUDE\s*$', line):
|
||||
elif undef_conditional_re.search(line):
|
||||
assert not file.is_conditional_include, f"Error: Amalgamated file '{file}' undefines SIMDJSON_CONDITIONAL_INCLUDE, which is not allowed. Only non-amalgamated files can undefine it. {rules}"
|
||||
assert self.in_conditional_include_block, f"Error: File '{file}' undefines SIMDJSON_CONDITIONAL_INCLUDE without having defined it first. Ensure proper define/undefine pairing. {rules}"
|
||||
self.write(f'/* undefining SIMDJSON_CONDITIONAL_INCLUDE */')
|
||||
@@ -554,7 +567,7 @@ def validate_implementations():
|
||||
|
||||
def read_version():
|
||||
with open(os.path.join(PROJECTPATH, 'include/simdjson/simdjson_version.h')) as f:
|
||||
return re.search(r'\d+\.\d+\.\d+', f.read()).group(0)
|
||||
return version_re.search(f.read()).group(0)
|
||||
|
||||
version = read_version()
|
||||
if not validate_implementations():
|
||||
|
||||
+6725
-85
File diff suppressed because it is too large
Load Diff
+22758
-96
File diff suppressed because it is too large
Load Diff
Binary file not shown.
@@ -18,4 +18,4 @@ simdjson_inline bool is_ascii(const simd8x64<uint8_t>& input);
|
||||
} // namespace SIMDJSON_IMPLEMENTATION
|
||||
} // namespace simdjson
|
||||
|
||||
#endif // SIMDJSON_SRC_GENERIC_DOM_PARSER_IMPLEMENTATION_H
|
||||
#endif // SIMDJSON_SRC_GENERIC_DOM_PARSER_IMPLEMENTATION_H
|
||||
|
||||
@@ -119,7 +119,7 @@ using namespace simd;
|
||||
simdjson_inline simd8<uint8_t> is_incomplete(const simd8<uint8_t> input) {
|
||||
// If the previous input's last 3 bytes match this, they're too short (they ended at EOF):
|
||||
// ... 1111____ 111_____ 11______
|
||||
#if SIMDJSON_IMPLEMENTATION_ICELAKE
|
||||
#if SIMDJSON_IMPLEMENTATION_ICELAKE || (SIMDJSON_IMPLEMENTATION_RVV_VLS && __riscv_v_fixed_vlen >= 512)
|
||||
static const uint8_t max_array[64] = {
|
||||
255, 255, 255, 255, 255, 255, 255, 255,
|
||||
255, 255, 255, 255, 255, 255, 255, 255,
|
||||
@@ -180,18 +180,18 @@ using namespace simd;
|
||||
|| (simd8x64<uint8_t>::NUM_CHUNKS == 4),
|
||||
"We support one, two or four chunks per 64-byte block.");
|
||||
SIMDJSON_IF_CONSTEXPR (simd8x64<uint8_t>::NUM_CHUNKS == 1) {
|
||||
this->check_utf8_bytes(input.chunks[0], this->prev_input_block);
|
||||
this->check_utf8_bytes(input.get<0>(), this->prev_input_block);
|
||||
} else SIMDJSON_IF_CONSTEXPR (simd8x64<uint8_t>::NUM_CHUNKS == 2) {
|
||||
this->check_utf8_bytes(input.chunks[0], this->prev_input_block);
|
||||
this->check_utf8_bytes(input.chunks[1], input.chunks[0]);
|
||||
this->check_utf8_bytes(input.get<0>(), this->prev_input_block);
|
||||
this->check_utf8_bytes(input.get<1>(), input.get<0>());
|
||||
} else SIMDJSON_IF_CONSTEXPR (simd8x64<uint8_t>::NUM_CHUNKS == 4) {
|
||||
this->check_utf8_bytes(input.chunks[0], this->prev_input_block);
|
||||
this->check_utf8_bytes(input.chunks[1], input.chunks[0]);
|
||||
this->check_utf8_bytes(input.chunks[2], input.chunks[1]);
|
||||
this->check_utf8_bytes(input.chunks[3], input.chunks[2]);
|
||||
this->check_utf8_bytes(input.get<0>(), this->prev_input_block);
|
||||
this->check_utf8_bytes(input.get<1>(), input.get<0>());
|
||||
this->check_utf8_bytes(input.get<2>(), input.get<1>());
|
||||
this->check_utf8_bytes(input.get<3>(), input.get<2>());
|
||||
}
|
||||
this->prev_incomplete = is_incomplete(input.chunks[simd8x64<uint8_t>::NUM_CHUNKS-1]);
|
||||
this->prev_input_block = input.chunks[simd8x64<uint8_t>::NUM_CHUNKS-1];
|
||||
this->prev_incomplete = is_incomplete(input.get<simd8x64<uint8_t>::NUM_CHUNKS-1>());
|
||||
this->prev_input_block = input.get<simd8x64<uint8_t>::NUM_CHUNKS-1>();
|
||||
}
|
||||
}
|
||||
// do not forget to call check_eof!
|
||||
@@ -206,4 +206,4 @@ using namespace simd;
|
||||
} // namespace SIMDJSON_IMPLEMENTATION
|
||||
} // namespace simdjson
|
||||
|
||||
#endif // SIMDJSON_SRC_GENERIC_STAGE1_UTF8_LOOKUP4_ALGORITHM_H
|
||||
#endif // SIMDJSON_SRC_GENERIC_STAGE1_UTF8_LOOKUP4_ALGORITHM_H
|
||||
|
||||
+19
-2
@@ -118,6 +118,17 @@ static const simdjson::lsx::implementation* get_lsx_singleton() {
|
||||
} // namespace simdjson
|
||||
#endif // SIMDJSON_IMPLEMENTATION_LSX
|
||||
|
||||
#if SIMDJSON_IMPLEMENTATION_RVV_VLS
|
||||
#include <simdjson/rvv-vls/implementation.h>
|
||||
namespace simdjson {
|
||||
namespace internal {
|
||||
static const simdjson::rvv_vls::implementation* get_rvv_vls_singleton() {
|
||||
static const simdjson::rvv_vls::implementation rvv_vls_singleton{};
|
||||
return &rvv_vls_singleton;
|
||||
}
|
||||
} // namespace internal
|
||||
} // namespace simdjson
|
||||
#endif // SIMDJSON_IMPLEMENTATION_RVV_VLS
|
||||
|
||||
#undef SIMDJSON_CONDITIONAL_INCLUDE
|
||||
|
||||
@@ -131,10 +142,10 @@ namespace internal {
|
||||
+ SIMDJSON_IMPLEMENTATION_HASWELL + SIMDJSON_IMPLEMENTATION_WESTMERE \
|
||||
+ SIMDJSON_IMPLEMENTATION_ARM64 + SIMDJSON_IMPLEMENTATION_PPC64 \
|
||||
+ SIMDJSON_IMPLEMENTATION_LSX + SIMDJSON_IMPLEMENTATION_LASX \
|
||||
+ SIMDJSON_IMPLEMENTATION_FALLBACK == 1)
|
||||
+ SIMDJSON_IMPLEMENTATION_RVV_VLS + SIMDJSON_IMPLEMENTATION_FALLBACK == 1)
|
||||
|
||||
#if SIMDJSON_SINGLE_IMPLEMENTATION
|
||||
static const implementation* get_single_implementation() {
|
||||
simdjson_really_inline static const implementation* get_single_implementation() {
|
||||
return
|
||||
#if SIMDJSON_IMPLEMENTATION_ICELAKE
|
||||
get_icelake_singleton();
|
||||
@@ -157,6 +168,9 @@ namespace internal {
|
||||
#if SIMDJSON_IMPLEMENTATION_LASX
|
||||
get_lasx_singleton();
|
||||
#endif
|
||||
#if SIMDJSON_IMPLEMENTATION_RVV_VLS
|
||||
get_rvv_vls_singleton();
|
||||
#endif
|
||||
#if SIMDJSON_IMPLEMENTATION_FALLBACK
|
||||
get_fallback_singleton();
|
||||
#endif
|
||||
@@ -217,6 +231,9 @@ static const std::initializer_list<const implementation *>& get_available_implem
|
||||
#if SIMDJSON_IMPLEMENTATION_LSX
|
||||
get_lsx_singleton(),
|
||||
#endif
|
||||
#if SIMDJSON_IMPLEMENTATION_RVV_VLS
|
||||
get_rvv_vls_singleton(),
|
||||
#endif
|
||||
#if SIMDJSON_IMPLEMENTATION_FALLBACK
|
||||
get_fallback_singleton(),
|
||||
#endif
|
||||
|
||||
@@ -239,6 +239,16 @@ static inline uint32_t detect_supported_architectures() {
|
||||
return host_isa;
|
||||
}
|
||||
|
||||
#elif SIMDJSON_IS_RISCV64
|
||||
|
||||
static inline uint32_t detect_supported_architectures() {
|
||||
uint32_t host_isa = instruction_set::DEFAULT;
|
||||
#if SIMDJSON_IS_RVV_VLS
|
||||
host_isa |= instruction_set::RVV_VLS;
|
||||
#endif
|
||||
return host_isa;
|
||||
}
|
||||
|
||||
#else // fallback
|
||||
|
||||
|
||||
|
||||
+132
@@ -0,0 +1,132 @@
|
||||
#ifndef SIMDJSON_SRC_RVV_VLS_CPP
|
||||
#define SIMDJSON_SRC_RVV_VLS_CPP
|
||||
|
||||
#ifndef SIMDJSON_CONDITIONAL_INCLUDE
|
||||
#include <base.h>
|
||||
#endif // SIMDJSON_CONDITIONAL_INCLUDE
|
||||
|
||||
#include <simdjson/rvv-vls.h>
|
||||
#include <simdjson/rvv-vls/implementation.h>
|
||||
|
||||
#include <simdjson/rvv-vls/begin.h>
|
||||
#include <simdjson/rvv-vls/simd.h>
|
||||
#include <generic/amalgamated.h>
|
||||
#include <generic/stage1/amalgamated.h>
|
||||
#include <generic/stage2/amalgamated.h>
|
||||
|
||||
//
|
||||
// Stage 1
|
||||
//
|
||||
namespace simdjson {
|
||||
namespace rvv_vls {
|
||||
|
||||
simdjson_warn_unused error_code implementation::create_dom_parser_implementation(
|
||||
size_t capacity,
|
||||
size_t max_depth,
|
||||
std::unique_ptr<internal::dom_parser_implementation>& dst
|
||||
) const noexcept {
|
||||
dst.reset( new (std::nothrow) dom_parser_implementation() );
|
||||
if (!dst) { return MEMALLOC; }
|
||||
if (auto err = dst->set_capacity(capacity))
|
||||
return err;
|
||||
if (auto err = dst->set_max_depth(max_depth))
|
||||
return err;
|
||||
return SUCCESS;
|
||||
}
|
||||
|
||||
namespace {
|
||||
|
||||
using namespace simd;
|
||||
|
||||
simdjson_inline json_character_block json_character_block::classify(const simd::simd8x64<uint8_t>& in) {
|
||||
static const uint8_t wsTable[16] = { ' ', 100, 100, 100, 17, 100, 113, 2, 100, '\t', '\n', 112, 100, '\r', 100, 100 };
|
||||
static const uint8_t opTable[16] = { 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, ':', '{', ',', '}', 0, 0 };
|
||||
vuint8_t vws = __riscv_vle8_v_u8m1(wsTable, 16);
|
||||
vuint8_t vop = __riscv_vle8_v_u8m1(opTable, 16);
|
||||
vuint8x64_t lo = __riscv_vand(in, 15, 64);
|
||||
vuint8x64_t curl = __riscv_vor(in, 0x20, 64);
|
||||
|
||||
#if __riscv_v_fixed_vlen == 128
|
||||
vboolx64_t mws = __riscv_vmseq(simdutf_vrgather_u8m1x4(vws, lo), in, 64);
|
||||
vboolx64_t mop = __riscv_vmseq(simdutf_vrgather_u8m1x4(vop, lo), curl, 64);
|
||||
#elif __riscv_v_fixed_vlen == 256
|
||||
vboolx64_t mws = __riscv_vmseq(simdutf_vrgather_u8m1x2(vws, lo), in, 64);
|
||||
vboolx64_t mop = __riscv_vmseq(simdutf_vrgather_u8m1x2(vop, lo), curl, 64);
|
||||
#else
|
||||
vboolx64_t mws = __riscv_vmseq(__riscv_vrgather(vws, lo, 64), in, 64);
|
||||
vboolx64_t mop = __riscv_vmseq(__riscv_vrgather(vop, lo, 64), curl, 64);
|
||||
#endif
|
||||
|
||||
return {
|
||||
__riscv_vmv_x(__riscv_vreinterpret_u64m1(mws)),
|
||||
__riscv_vmv_x(__riscv_vreinterpret_u64m1(mop))
|
||||
};
|
||||
}
|
||||
|
||||
simdjson_inline bool is_ascii(const simd8x64<uint8_t>& input) {
|
||||
return input.is_ascii();
|
||||
}
|
||||
|
||||
simdjson_inline simd8<uint8_t> must_be_2_3_continuation(const simd8<uint8_t> prev2, const simd8<uint8_t> prev3) {
|
||||
simd8<uint8_t> is_third_byte = prev2.saturating_sub(0xe0u-0x80); // Only 111_____ will be >= 0x80
|
||||
simd8<uint8_t> is_fourth_byte = prev3.saturating_sub(0xf0u-0x80); // Only 1111____ will be >= 0x80
|
||||
return is_third_byte | is_fourth_byte;
|
||||
}
|
||||
|
||||
} // unnamed namespace
|
||||
} // namespace rvv_vls
|
||||
} // namespace simdjson
|
||||
|
||||
//
|
||||
// Stage 2
|
||||
//
|
||||
|
||||
//
|
||||
// Implementation-specific overrides
|
||||
//
|
||||
namespace simdjson {
|
||||
namespace rvv_vls {
|
||||
|
||||
simdjson_warn_unused error_code implementation::minify(const uint8_t *buf, size_t len, uint8_t *dst, size_t &dst_len) const noexcept {
|
||||
return rvv_vls::stage1::json_minifier::minify<64>(buf, len, dst, dst_len);
|
||||
}
|
||||
|
||||
simdjson_warn_unused error_code dom_parser_implementation::stage1(const uint8_t *_buf, size_t _len, stage1_mode streaming) noexcept {
|
||||
this->buf = _buf;
|
||||
this->len = _len;
|
||||
return rvv_vls::stage1::json_structural_indexer::index<64>(buf, len, *this, streaming);
|
||||
}
|
||||
|
||||
simdjson_warn_unused bool implementation::validate_utf8(const char *buf, size_t len) const noexcept {
|
||||
return rvv_vls::stage1::generic_validate_utf8(buf,len);
|
||||
}
|
||||
|
||||
simdjson_warn_unused error_code dom_parser_implementation::stage2(dom::document &_doc) noexcept {
|
||||
return stage2::tape_builder::parse_document<false>(*this, _doc);
|
||||
}
|
||||
|
||||
simdjson_warn_unused error_code dom_parser_implementation::stage2_next(dom::document &_doc) noexcept {
|
||||
return stage2::tape_builder::parse_document<true>(*this, _doc);
|
||||
}
|
||||
|
||||
SIMDJSON_NO_SANITIZE_MEMORY
|
||||
simdjson_warn_unused uint8_t *dom_parser_implementation::parse_string(const uint8_t *src, uint8_t *dst, bool allow_replacement) const noexcept {
|
||||
return rvv_vls::stringparsing::parse_string(src, dst, allow_replacement);
|
||||
}
|
||||
|
||||
simdjson_warn_unused uint8_t *dom_parser_implementation::parse_wobbly_string(const uint8_t *src, uint8_t *dst) const noexcept {
|
||||
return rvv_vls::stringparsing::parse_wobbly_string(src, dst);
|
||||
}
|
||||
|
||||
simdjson_warn_unused error_code dom_parser_implementation::parse(const uint8_t *_buf, size_t _len, dom::document &_doc) noexcept {
|
||||
auto error = stage1(_buf, _len, stage1_mode::regular);
|
||||
if (error) { return error; }
|
||||
return stage2(_doc);
|
||||
}
|
||||
|
||||
} // namespace rvv_vls
|
||||
} // namespace simdjson
|
||||
|
||||
#include <simdjson/rvv-vls/end.h>
|
||||
|
||||
#endif // SIMDJSON_SRC_RVV_VLS_CPP
|
||||
@@ -41,6 +41,9 @@ SIMDJSON_PUSH_DISABLE_UNUSED_WARNINGS
|
||||
#if SIMDJSON_IMPLEMENTATION_LSX
|
||||
#include <lsx.cpp>
|
||||
#endif
|
||||
#if SIMDJSON_IMPLEMENTATION_RVV_VLS
|
||||
#include <rvv-vls.cpp>
|
||||
#endif
|
||||
#if SIMDJSON_IMPLEMENTATION_FALLBACK
|
||||
#include <fallback.cpp>
|
||||
#endif
|
||||
|
||||
@@ -25,3 +25,8 @@ if (SIMDJSON_EXCEPTIONS)
|
||||
add_dual_compile_test(dangling_parser_parse_padstring)
|
||||
add_dual_compile_test(unsafe_parse_many)
|
||||
endif()
|
||||
|
||||
if(NOT BUILD_SHARED_LIBS)
|
||||
# We only check that it builds
|
||||
add_subdirectory(multiple_include)
|
||||
endif()
|
||||
@@ -0,0 +1,5 @@
|
||||
add_library(mylib mylib.cpp)
|
||||
target_link_libraries(mylib PUBLIC simdjson::simdjson)
|
||||
target_include_directories(mylib PUBLIC ${CMAKE_CURRENT_SOURCE_DIR})
|
||||
add_executable(myexe main.cpp)
|
||||
target_link_libraries(myexe PRIVATE mylib)
|
||||
@@ -0,0 +1,22 @@
|
||||
#include "mylib.h"
|
||||
#include <simdjson.h>
|
||||
#include <iostream>
|
||||
#include <memory>
|
||||
|
||||
int main() {
|
||||
simdjson::padded_string json = R"([ 1, 2, 3, 4 ])"_padded;
|
||||
simdjson::padded_string minified = minify_json(json);
|
||||
std::cout << "Minified: " << std::string_view(minified) << std::endl;
|
||||
|
||||
// Also directly use minify
|
||||
size_t length = json.size();
|
||||
std::unique_ptr<char[]> buffer{new char[length]};
|
||||
size_t new_length{};
|
||||
auto error = simdjson::minify(json.data(), length, buffer.get(), new_length);
|
||||
if (error) {
|
||||
std::cerr << "Error: " << simdjson::error_message(error) << std::endl;
|
||||
} else {
|
||||
std::cout << "Direct minified: " << std::string(buffer.get(), new_length) << std::endl;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,14 @@
|
||||
#include "mylib.h"
|
||||
#include <simdjson.h>
|
||||
#include <memory>
|
||||
|
||||
simdjson::padded_string minify_json(const simdjson::padded_string& json) {
|
||||
size_t length = json.size();
|
||||
std::unique_ptr<char[]> buffer{new char[length]};
|
||||
size_t new_length{};
|
||||
auto error = simdjson::minify(json.data(), length, buffer.get(), new_length);
|
||||
if (error) {
|
||||
return simdjson::padded_string();
|
||||
}
|
||||
return simdjson::padded_string(std::string_view(buffer.get(), new_length));
|
||||
}
|
||||
@@ -0,0 +1,9 @@
|
||||
#ifndef MYLIB_H
|
||||
#define MYLIB_H
|
||||
|
||||
#include <simdjson.h>
|
||||
#include <string>
|
||||
|
||||
simdjson::padded_string minify_json(const simdjson::padded_string& json);
|
||||
|
||||
#endif // MYLIB_H
|
||||
@@ -79,7 +79,7 @@ if (BASH AND (NOT WIN32) AND SIMDJSON_BASH AND (TARGET json2json)) # The scripts
|
||||
$<TARGET_FILE:checkimplementation>
|
||||
)
|
||||
endif()
|
||||
if(CMAKE_HOST_SYSTEM_PROCESSOR STREQUAL x86_64 OR CMAKE_HOST_SYSTEM_PROCESSOR STREQUAL amd64)
|
||||
if((CMAKE_HOST_SYSTEM_PROCESSOR STREQUAL x86_64 OR CMAKE_HOST_SYSTEM_PROCESSOR STREQUAL amd64) AND NOT DEFINED CMAKE_CROSSCOMPILING_EMULATOR)
|
||||
add_test(
|
||||
NAME simdjson_force_implementation_error
|
||||
COMMAND
|
||||
|
||||
Reference in New Issue
Block a user