Compare commits
79 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 8e85352df4 | |||
| fc57c09cf0 | |||
| 2058b47dfe | |||
| b0486c7fa2 | |||
| feb1e7feb6 | |||
| 2c7fbc1538 | |||
| 8b69401d8a | |||
| f504e57e7a | |||
| 135c173053 | |||
| b4ed3a99a9 | |||
| d8e1b36c88 | |||
| c249a1b456 | |||
| 9c4b793c90 | |||
| dd92a8414c | |||
| 835bdba123 | |||
| ad3cd71ca2 | |||
| 860f7e0458 | |||
| 980f2ad3af | |||
| 7ad9fe63a6 | |||
| 7987418b1f | |||
| 5e871f6724 | |||
| 5d16fd5f31 | |||
| 4e9ff03af5 | |||
| aa7489060a | |||
| ae32422891 | |||
| 667d0ed3c7 | |||
| b1c31b428d | |||
| 56ac56ba32 | |||
| 19549c60ec | |||
| 16e99f229b | |||
| 21342a4142 | |||
| d0e841d3e9 | |||
| a962652ec3 | |||
| 77d73b068a | |||
| 19ff7a572d | |||
| 235dbc5369 | |||
| 2b9c8977af | |||
| 3d87bd4abc | |||
| a60c0d1e39 | |||
| 86bfbaada7 | |||
| ccac6403d9 | |||
| aade58c3dc | |||
| 3319815e25 | |||
| a24f845bd7 | |||
| 212e2d5857 | |||
| 547e156e33 | |||
| 98a45f7229 | |||
| 8589509d1e | |||
| 2fce4a843d | |||
| 62803512e4 | |||
| 76d9dee854 | |||
| bf15f21b0b | |||
| 32b301893c | |||
| 7f1531a1f9 | |||
| 0a3b555ff7 | |||
| 114d45ad54 | |||
| 0112be86b0 | |||
| bf52d8198b | |||
| 58c92d6d82 | |||
| 781a7d6c89 | |||
| 3d0de709a8 | |||
| 49b86721b4 | |||
| def2b6efd2 | |||
| 87a186fbf1 | |||
| 81f10a01b7 | |||
| 36ed7ab48a | |||
| c3d1d62dfe | |||
| 67821cb6fd | |||
| 3ac287ba3d | |||
| ec352430a0 | |||
| d326f2ce9f | |||
| ca42a49fba | |||
| 8a9daeb0ad | |||
| 9c5a88f1f3 | |||
| a7f8fb71c5 | |||
| a553db4c67 | |||
| 1fa1af8c15 | |||
| 403b8bfb91 | |||
| 76f45a0c4b |
@@ -1,316 +0,0 @@
|
||||
version: 2.1
|
||||
|
||||
|
||||
# We constantly run out of memory so please do not use parallelism (-j, -j4).
|
||||
|
||||
# Reusable image / compiler definitions
|
||||
executors:
|
||||
gcc8:
|
||||
docker:
|
||||
- image: conanio/gcc8
|
||||
environment:
|
||||
CXX: g++-8
|
||||
CC: gcc-8
|
||||
CMAKE_BUILD_FLAGS:
|
||||
CTEST_FLAGS: --output-on-failure
|
||||
|
||||
gcc9:
|
||||
docker:
|
||||
- image: conanio/gcc9
|
||||
environment:
|
||||
CXX: g++-9
|
||||
CC: gcc-9
|
||||
CMAKE_BUILD_FLAGS:
|
||||
CTEST_FLAGS: --output-on-failure
|
||||
|
||||
gcc10:
|
||||
docker:
|
||||
- image: conanio/gcc10
|
||||
environment:
|
||||
CXX: g++-10
|
||||
CC: gcc-10
|
||||
CMAKE_BUILD_FLAGS:
|
||||
CTEST_FLAGS: --output-on-failure
|
||||
|
||||
clang10:
|
||||
docker:
|
||||
- image: conanio/clang10
|
||||
environment:
|
||||
CXX: clang++-10
|
||||
CC: clang-10
|
||||
CMAKE_BUILD_FLAGS:
|
||||
CTEST_FLAGS: --output-on-failure
|
||||
|
||||
clang9:
|
||||
docker:
|
||||
- image: conanio/clang9
|
||||
environment:
|
||||
CXX: clang++-9
|
||||
CC: clang-9
|
||||
CMAKE_BUILD_FLAGS:
|
||||
CTEST_FLAGS: --output-on-failure
|
||||
|
||||
clang6:
|
||||
docker:
|
||||
- image: conanio/clang60
|
||||
environment:
|
||||
CXX: clang++-6.0
|
||||
CC: clang-6.0
|
||||
CMAKE_BUILD_FLAGS:
|
||||
CTEST_FLAGS: --output-on-failure
|
||||
|
||||
# Reusable test commands (and initializer for clang 6)
|
||||
commands:
|
||||
dependency_restore:
|
||||
steps:
|
||||
- restore_cache:
|
||||
keys:
|
||||
- cmake-cache-{{ checksum "dependencies/CMakeLists.txt" }}
|
||||
|
||||
dependency_cache:
|
||||
steps:
|
||||
- save_cache:
|
||||
key: cmake-cache-{{ checksum "dependencies/CMakeLists.txt" }}
|
||||
paths:
|
||||
- dependencies/.cache
|
||||
|
||||
install_cmake:
|
||||
steps:
|
||||
- run: apt-get update -qq
|
||||
- run: apt-get install -y cmake
|
||||
|
||||
cmake_prep:
|
||||
steps:
|
||||
- checkout
|
||||
- run: mkdir -p build
|
||||
|
||||
cmake_build_cache:
|
||||
steps:
|
||||
- cmake_prep
|
||||
- dependency_restore
|
||||
- run: cmake -DSIMDJSON_DEVELOPER_MODE=ON $CMAKE_FLAGS -DCMAKE_INSTALL_PREFIX:PATH=destination -B build .
|
||||
- dependency_cache # dependencies are produced in the configure step
|
||||
|
||||
cmake_build:
|
||||
steps:
|
||||
- cmake_build_cache
|
||||
- run: cmake --build build
|
||||
|
||||
cmake_test:
|
||||
steps:
|
||||
- cmake_build
|
||||
- run: |
|
||||
cd build &&
|
||||
tools/json2json -h &&
|
||||
ctest $CTEST_FLAGS -L acceptance &&
|
||||
ctest $CTEST_FLAGS -LE acceptance -LE explicitonly
|
||||
|
||||
cmake_assert_test:
|
||||
steps:
|
||||
- run: |
|
||||
cd build &&
|
||||
tools/json2json -h &&
|
||||
ctest $CTEST_FLAGS -L assert
|
||||
|
||||
cmake_test_all:
|
||||
steps:
|
||||
- cmake_build
|
||||
- run: |
|
||||
cd build &&
|
||||
tools/json2json -h &&
|
||||
ctest $CTEST_FLAGS -DSIMDJSON_IMPLEMENTATION="haswell;westmere;fallback" -L acceptance -LE per_implementation &&
|
||||
SIMDJSON_FORCE_IMPLEMENTATION=haswell ctest $CTEST_FLAGS -L per_implementation -LE explicitonly &&
|
||||
SIMDJSON_FORCE_IMPLEMENTATION=westmere ctest $CTEST_FLAGS -L per_implementation -LE explicitonly &&
|
||||
SIMDJSON_FORCE_IMPLEMENTATION=fallback ctest $CTEST_FLAGS -L per_implementation -LE explicitonly &&
|
||||
ctest $CTEST_FLAGS -LE "acceptance|per_implementation" # Everything we haven't run yet, run now.
|
||||
|
||||
|
||||
cmake_perftest:
|
||||
steps:
|
||||
- cmake_build_cache
|
||||
- run: |
|
||||
cmake -DSIMDJSON_ENABLE_DOM_CHECKPERF=ON --build build --target checkperf &&
|
||||
cd build &&
|
||||
ctest --output-on-failure -R checkperf
|
||||
|
||||
# we not only want cmake to build and run tests, but we want also a successful installation from which we can build, link and run programs
|
||||
cmake_install_test: # this version builds, install, test and then verify from the installation
|
||||
steps:
|
||||
- run: cd build && make install
|
||||
- run: echo -e '#include <simdjson.h>\nint main(int argc,char**argv) {simdjson::dom::parser parser;simdjson::dom::element tweets = parser.load(argv[1]); }' > tmp.cpp && c++ -Ibuild/destination/include -Lbuild/destination/lib -std=c++17 -Wl,-rpath,build/destination/lib -o linkandrun tmp.cpp -lsimdjson && ./linkandrun jsonexamples/twitter.json
|
||||
|
||||
cmake_installed_test_cxx20: # assuming that it was installed, this tries to build using C++20
|
||||
steps:
|
||||
- run: echo -e '#include <simdjson.h>\nint main(int argc,char**argv) {simdjson::dom::parser parser;simdjson::dom::element tweets = parser.load(argv[1]); }' > tmp.cpp && c++ -Ibuild/destination/include -Lbuild/destination/lib -std=c++20 -Wl,-rpath,build/destination/lib -o linkandrun tmp.cpp -lsimdjson && ./linkandrun jsonexamples/twitter.json
|
||||
|
||||
jobs:
|
||||
|
||||
# static
|
||||
justlib-gcc10:
|
||||
description: Build just the library, install it and do a basic test
|
||||
executor: gcc10
|
||||
environment: { CMAKE_FLAGS: -DSIMDJSON_JUST_LIBRARY=ON }
|
||||
steps: [ cmake_build, cmake_install_test, cmake_installed_test_cxx20 ]
|
||||
assert-gcc10:
|
||||
description: Build the library with asserts on, install it and run tests
|
||||
executor: gcc10
|
||||
environment: { CMAKE_FLAGS: -DSIMDJSON_GOOGLE_BENCHMARKS=OFF -DCMAKE_CXX_FLAGS_RELEASE=-O3 }
|
||||
steps: [ cmake_test, cmake_assert_test ]
|
||||
assert-clang10:
|
||||
description: Build just the library, install it and do a basic test
|
||||
executor: clang10
|
||||
environment: { CMAKE_FLAGS: -DSIMDJSON_GOOGLE_BENCHMARKS=OFF -DCMAKE_CXX_FLAGS_RELEASE=-O3 }
|
||||
steps: [ cmake_test, cmake_assert_test ]
|
||||
gcc10-perftest:
|
||||
description: Build and run performance tests on GCC 10 and AVX 2 with a cmake static build, this test performance regression
|
||||
executor: gcc10
|
||||
environment: { CMAKE_FLAGS: -DSIMDJSON_GOOGLE_BENCHMARKS=OFF -DBUILD_SHARED_LIBS=OFF }
|
||||
steps: [ cmake_perftest ]
|
||||
gcc10:
|
||||
description: Build and run tests on GCC 10 and AVX 2 with a cmake static build
|
||||
executor: gcc10
|
||||
environment: { CMAKE_FLAGS: -DSIMDJSON_GOOGLE_BENCHMARKS=ON -DBUILD_SHARED_LIBS=OFF }
|
||||
steps: [ cmake_test, cmake_install_test, cmake_installed_test_cxx20 ]
|
||||
clang6:
|
||||
description: Build and run tests on clang 6 and AVX 2 with a cmake static build
|
||||
executor: clang6
|
||||
environment: { CMAKE_FLAGS: -DSIMDJSON_GOOGLE_BENCHMARKS=ON -DBUILD_SHARED_LIBS=OFF }
|
||||
steps: [ cmake_test, cmake_install_test ]
|
||||
clang10:
|
||||
description: Build and run tests on clang 10 and AVX 2 with a cmake static build
|
||||
executor: clang10
|
||||
environment: { CMAKE_FLAGS: -DSIMDJSON_GOOGLE_BENCHMARKS=ON -DBUILD_SHARED_LIBS=OFF }
|
||||
steps: [ cmake_test, cmake_install_test, cmake_installed_test_cxx20 ]
|
||||
# libcpp
|
||||
libcpp-clang10:
|
||||
description: Build and run tests on clang 10 and AVX 2 with a cmake static build and libc++
|
||||
executor: clang10
|
||||
environment: { CMAKE_FLAGS: -DSIMDJSON_USE_LIBCPP=ON -DBUILD_SHARED_LIBS=OFF }
|
||||
steps: [ cmake_test, cmake_install_test, cmake_installed_test_cxx20 ]
|
||||
# sanitize
|
||||
sanitize-gcc10:
|
||||
description: Build and run tests on GCC 10 and AVX 2 with a cmake sanitize build
|
||||
executor: gcc10
|
||||
environment: { CMAKE_FLAGS: -DCMAKE_BUILD_TYPE=Debug -DBUILD_SHARED_LIBS=ON -DSIMDJSON_SANITIZE=ON, CTEST_FLAGS: --output-on-failure -LE explicitonly }
|
||||
steps: [ cmake_test ]
|
||||
sanitize-clang10:
|
||||
description: Build and run tests on clang 10 and AVX 2 with a cmake sanitize build
|
||||
executor: clang10
|
||||
environment: { CMAKE_FLAGS: -DBUILD_SHARED_LIBS=ON -DSIMDJSON_NO_FORCE_INLINING=ON -DSIMDJSON_SANITIZE=ON, CTEST_FLAGS: --output-on-failure -LE explicitonly }
|
||||
steps: [ cmake_test ]
|
||||
threadsanitize-gcc10:
|
||||
description: Build and run tests on GCC 10 and AVX 2 with a cmake sanitize build
|
||||
executor: gcc10
|
||||
environment: { CMAKE_FLAGS: -DBUILD_SHARED_LIBS=ON -DSIMDJSON_SANITIZE_THREADS=ON, CTEST_FLAGS: --output-on-failure -LE explicitonly }
|
||||
steps: [ cmake_test ]
|
||||
threadsanitize-clang10:
|
||||
description: Build and run tests on clang 10 and AVX 2 with a cmake sanitize build
|
||||
executor: clang10
|
||||
environment: { CMAKE_FLAGS: -DBUILD_SHARED_LIBS=ON -DSIMDJSON_NO_FORCE_INLINING=ON -DSIMDJSON_SANITIZE_THREADS=ON, CTEST_FLAGS: --output-on-failure -LE explicitonly }
|
||||
steps: [ cmake_test ]
|
||||
# dynamic
|
||||
dynamic-gcc10:
|
||||
description: Build and run tests on GCC 10 and AVX 2 with a cmake dynamic build
|
||||
executor: gcc10
|
||||
environment: { CMAKE_FLAGS: -DBUILD_SHARED_LIBS=ON }
|
||||
steps: [ cmake_test, cmake_install_test ]
|
||||
dynamic-clang10:
|
||||
description: Build and run tests on clang 10 and AVX 2 with a cmake dynamic build
|
||||
executor: clang10
|
||||
environment: { CMAKE_FLAGS: -DBUILD_SHARED_LIBS=ON }
|
||||
steps: [ cmake_test, cmake_install_test ]
|
||||
|
||||
# unthreaded
|
||||
unthreaded-gcc10:
|
||||
description: Build and run tests on GCC 10 and AVX 2 *without* threads
|
||||
executor: gcc10
|
||||
environment: { CMAKE_FLAGS: -DSIMDJSON_ENABLE_THREADS=OFF }
|
||||
steps: [ cmake_test, cmake_install_test ]
|
||||
unthreaded-clang10:
|
||||
description: Build and run tests on Clang 10 and AVX 2 *without* threads
|
||||
executor: clang10
|
||||
environment: { CMAKE_FLAGS: -DSIMDJSON_ENABLE_THREADS=OFF }
|
||||
steps: [ cmake_test, cmake_install_test ]
|
||||
|
||||
# noexcept
|
||||
noexcept-gcc10:
|
||||
description: Build and run tests on GCC 10 and AVX 2 with exceptions off
|
||||
executor: gcc10
|
||||
environment: { CMAKE_FLAGS: -DSIMDJSON_EXCEPTIONS=OFF }
|
||||
steps: [ cmake_test, cmake_install_test ]
|
||||
noexcept-clang10:
|
||||
description: Build and run tests on Clang 10 and AVX 2 with exceptions off
|
||||
executor: clang10
|
||||
environment: { CMAKE_FLAGS: -DSIMDJSON_EXCEPTIONS=OFF }
|
||||
steps: [ cmake_test, cmake_install_test ]
|
||||
|
||||
#
|
||||
# Misc.
|
||||
#
|
||||
|
||||
# make (test and checkperf)
|
||||
arch-haswell-gcc10:
|
||||
description: Build, run tests and check performance on GCC 10 with -march=haswell
|
||||
executor: gcc10
|
||||
environment: { CXXFLAGS: -march=haswell }
|
||||
steps: [ cmake_test ]
|
||||
arch-nehalem-gcc10:
|
||||
description: Build, run tests and check performance on GCC 10 with -march=nehalem
|
||||
executor: gcc10
|
||||
environment: { CXXFLAGS: -march=nehalem }
|
||||
steps: [ cmake_test ]
|
||||
sanitize-haswell-gcc10:
|
||||
description: Build and run tests on GCC 10 and AVX 2 with a cmake sanitize build
|
||||
executor: gcc10
|
||||
environment: { CXXFLAGS: -march=haswell, CMAKE_FLAGS: -DCMAKE_BUILD_TYPE=Debug -DBUILD_SHARED_LIBS=ON -DSIMDJSON_SANITIZE=ON, CTEST_FLAGS: --output-on-failure -LE explicitonly }
|
||||
steps: [ cmake_test ]
|
||||
sanitize-haswell-clang10:
|
||||
description: Build and run tests on clang 10 and AVX 2 with a cmake sanitize build
|
||||
executor: clang10
|
||||
environment: { CXXFLAGS: -march=haswell, CMAKE_FLAGS: -DBUILD_SHARED_LIBS=ON -DSIMDJSON_NO_FORCE_INLINING=ON -DSIMDJSON_SANITIZE=ON, CTEST_FLAGS: --output-on-failure -LE explicitonly }
|
||||
steps: [ cmake_test ]
|
||||
|
||||
workflows:
|
||||
version: 2.1
|
||||
build_and_test:
|
||||
jobs:
|
||||
# full multi-implementation tests
|
||||
#- gcc7 tested on GitHub actions
|
||||
- gcc10 # do not delete this as it tests our performance
|
||||
- clang6
|
||||
#- clang10 # this gets tested a lot below
|
||||
|
||||
# libc++
|
||||
- libcpp-clang10
|
||||
|
||||
# full single-implementation tests
|
||||
- sanitize-gcc10
|
||||
- sanitize-clang10
|
||||
- threadsanitize-gcc10
|
||||
- threadsanitize-clang10
|
||||
- dynamic-gcc10
|
||||
- dynamic-clang10
|
||||
- unthreaded-gcc10
|
||||
- unthreaded-clang10
|
||||
|
||||
# no exceptions
|
||||
- noexcept-gcc10
|
||||
- noexcept-clang10
|
||||
|
||||
# quicker make single-implementation tests
|
||||
- arch-haswell-gcc10
|
||||
- arch-nehalem-gcc10
|
||||
|
||||
|
||||
# sanitized single-implementation tests
|
||||
- sanitize-haswell-gcc10
|
||||
- sanitize-haswell-clang10
|
||||
|
||||
# testing "just the library"
|
||||
- justlib-gcc10
|
||||
|
||||
# testing asserts
|
||||
- assert-gcc10
|
||||
- assert-clang10
|
||||
|
||||
# TODO add windows: https://circleci.com/docs/2.0/configuration-reference/#windows
|
||||
@@ -38,7 +38,7 @@ If we cannot reproduce the issue, then we cannot address it. Note that a stack t
|
||||
|
||||
It should be possible to trigger the bug by using solely simdjson with our default build setup. If you can only observe the bug within some specific context, with some other software, please reduce the issue first.
|
||||
|
||||
**simjson release**
|
||||
**simdjson release**
|
||||
|
||||
Unless you plan to contribute to simdjson, you should only work from releases. Please be mindful that our main branch may have additional features, bugs and documentation items.
|
||||
|
||||
|
||||
@@ -6,6 +6,7 @@ Description
|
||||
|
||||
Type of change
|
||||
- [ ] Bug fix
|
||||
- [ ] Optimization
|
||||
- [ ] New feature
|
||||
- [ ] Refactor / cleanup
|
||||
- [ ] Documentation / tests
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
name: Ubuntu ppc64le (GCC 11)
|
||||
name: Ubuntu aarch64 (GCC 13)
|
||||
|
||||
on:
|
||||
push:
|
||||
|
||||
@@ -2,7 +2,13 @@ name: Doxygen GitHub Pages
|
||||
|
||||
on:
|
||||
release:
|
||||
types: [created]
|
||||
# Trigger when a release object is created and when it's published.
|
||||
# Some GitHub flows create a release object then publish it later; include both.
|
||||
types: [created, published]
|
||||
# Also trigger on tag creation pushes so releasing via Git tags still runs the workflow
|
||||
push:
|
||||
tags:
|
||||
- "v*" # common release tag pattern like v1.2.3
|
||||
# Allows you to run this workflow manually from the Actions tab
|
||||
workflow_dispatch:
|
||||
|
||||
@@ -27,7 +33,7 @@ jobs:
|
||||
- name: Generate Doxygen Documentation
|
||||
run: doxygen
|
||||
- name: Deploy to GitHub Pages
|
||||
uses: peaceiris/actions-gh-pages@v3
|
||||
uses: peaceiris/actions-gh-pages@v4
|
||||
with:
|
||||
github_token: ${{ secrets.GITHUB_TOKEN }}
|
||||
publish_dir: doc/api/html
|
||||
|
||||
@@ -1,9 +1,18 @@
|
||||
cmake_minimum_required(VERSION 3.14)
|
||||
|
||||
# Build performance optimizations
|
||||
set(CMAKE_EXPORT_COMPILE_COMMANDS ON CACHE BOOL "Export compile commands for faster IDE integration")
|
||||
set_property(GLOBAL PROPERTY USE_FOLDERS ON)
|
||||
|
||||
# Enable parallel compilation on MSVC
|
||||
if(MSVC)
|
||||
add_compile_options(/MP)
|
||||
endif()
|
||||
|
||||
project(
|
||||
simdjson
|
||||
# The version number is modified by tools/release.py
|
||||
VERSION 4.0.7
|
||||
VERSION 4.2.4
|
||||
DESCRIPTION "Parsing gigabytes of JSON per second"
|
||||
HOMEPAGE_URL "https://simdjson.org/"
|
||||
LANGUAGES CXX C
|
||||
@@ -20,8 +29,8 @@ string(
|
||||
# ---- Options, variables ----
|
||||
|
||||
# These version numbers are modified by tools/release.py
|
||||
set(SIMDJSON_LIB_VERSION "27.0.0" CACHE STRING "simdjson library version")
|
||||
set(SIMDJSON_LIB_SOVERSION "27" CACHE STRING "simdjson library soversion")
|
||||
set(SIMDJSON_LIB_VERSION "29.0.0" CACHE STRING "simdjson library version")
|
||||
set(SIMDJSON_LIB_SOVERSION "29" CACHE STRING "simdjson library soversion")
|
||||
|
||||
option(SIMDJSON_BUILD_STATIC_LIB "Build simdjson_static library along with simdjson (only makes sense if BUILD_SHARED_LIBS=ON)" OFF)
|
||||
if(SIMDJSON_BUILD_STATIC_LIB AND NOT BUILD_SHARED_LIBS)
|
||||
@@ -83,10 +92,36 @@ add_library(simdjson ${SIMDJSON_SOURCES})
|
||||
add_library(simdjson::simdjson ALIAS simdjson)
|
||||
set(SIMDJSON_LIBRARIES simdjson)
|
||||
|
||||
# Enable precompiled headers for faster builds
|
||||
if(CMAKE_VERSION VERSION_GREATER_EQUAL "3.16")
|
||||
target_precompile_headers(simdjson PRIVATE
|
||||
<algorithm>
|
||||
<array>
|
||||
<atomic>
|
||||
<bit>
|
||||
<cassert>
|
||||
<cctype>
|
||||
<cerrno>
|
||||
<cstddef>
|
||||
<cstdint>
|
||||
<cstdlib>
|
||||
<cstring>
|
||||
<memory>
|
||||
<string>
|
||||
<utility>
|
||||
<vector>
|
||||
)
|
||||
endif()
|
||||
|
||||
if(SIMDJSON_BUILD_STATIC_LIB)
|
||||
add_library(simdjson_static STATIC ${SIMDJSON_SOURCES})
|
||||
add_library(simdjson::simdjson_static ALIAS simdjson_static)
|
||||
list(APPEND SIMDJSON_LIBRARIES simdjson_static)
|
||||
|
||||
# Reuse precompiled headers for static library
|
||||
if(CMAKE_VERSION VERSION_GREATER_EQUAL "3.16")
|
||||
target_precompile_headers(simdjson_static REUSE_FROM simdjson)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
set_target_properties(
|
||||
@@ -112,6 +147,14 @@ simdjson_add_props(
|
||||
PRIVATE "$<BUILD_INTERFACE:${PROJECT_SOURCE_DIR}/src>"
|
||||
)
|
||||
|
||||
# Optimize linker settings for faster builds
|
||||
if(MSVC)
|
||||
target_link_options(simdjson PRIVATE /INCREMENTAL)
|
||||
if(CMAKE_BUILD_TYPE STREQUAL "Debug")
|
||||
target_link_options(simdjson PRIVATE /DEBUG:FASTLINK)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
if(SIMDJSON_STATIC_REFLECTION)
|
||||
# We would like to require C++26, but no compiler supports that!
|
||||
# This is a hack:
|
||||
@@ -140,24 +183,6 @@ if(SIMDJSON_MINUS_ZERO_AS_FLOAT)
|
||||
simdjson_add_props(target_compile_definitions PRIVATE SIMDJSON_MINUS_ZERO_AS_FLOAT=1)
|
||||
endif(SIMDJSON_MINUS_ZERO_AS_FLOAT)
|
||||
|
||||
if(CMAKE_SYSTEM_PROCESSOR MATCHES "^(loongarch64)$")
|
||||
option(SIMDJSON_PREFER_LSX "Prefer LoongArch SX" ON)
|
||||
include(CheckCXXCompilerFlag)
|
||||
check_cxx_compiler_flag(-mlasx COMPILER_SUPPORTS_LASX)
|
||||
check_cxx_compiler_flag(-mlsx COMPILER_SUPPORTS_LSX)
|
||||
if(COMPILER_SUPPORTS_LASX AND NOT SIMDJSON_PREFER_LSX)
|
||||
simdjson_add_props(
|
||||
target_compile_options PRIVATE
|
||||
-mlasx
|
||||
)
|
||||
elseif(COMPILER_SUPPORTS_LSX)
|
||||
simdjson_add_props(
|
||||
target_compile_options PRIVATE
|
||||
-mlsx
|
||||
)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
# GCC and Clang have horrendous Debug builds when using SIMD.
|
||||
# A common fix is to use '-Og' instead.
|
||||
# bug https://gcc.gnu.org/bugzilla/show_bug.cgi?id=54412
|
||||
@@ -171,7 +196,12 @@ if(
|
||||
target_compile_options PRIVATE
|
||||
$<$<CONFIG:DEBUG>:-Og>
|
||||
)
|
||||
endif()
|
||||
# We still want to enable development checks in Debug mode
|
||||
simdjson_add_props(
|
||||
target_compile_definitions PUBLIC
|
||||
SIMDJSON_DEVELOPMENT_CHECKS
|
||||
)
|
||||
endif()
|
||||
|
||||
if(SIMDJSON_ENABLE_THREADS)
|
||||
find_package(Threads REQUIRED)
|
||||
|
||||
@@ -38,7 +38,7 @@ PROJECT_NAME = simdjson
|
||||
# could be handy for archiving the generated documentation or if some version
|
||||
# control system is used.
|
||||
|
||||
PROJECT_NUMBER = "4.0.7"
|
||||
PROJECT_NUMBER = "4.2.4"
|
||||
|
||||
# Using the PROJECT_BRIEF tag one can provide an optional one line description
|
||||
# for a project that appears at the top of each page and should give viewer a
|
||||
|
||||
@@ -110,6 +110,24 @@ workflows used by simdjson.
|
||||
Directory Structure and Source
|
||||
------------------------------
|
||||
|
||||
Before diving into the directory structure, here are key concepts used in the codebase:
|
||||
|
||||
- **Amalgamated File**: A file that is conditionally included in the amalgamation process. These are wrapped in `#ifndef SIMDJSON_CONDITIONAL_INCLUDE` blocks and are included based on the target implementation (e.g., ARM64, x86). They include implementation-specific files (e.g., `arm64.h`) and generic files (e.g., under `generic/`). Amalgamated files have associated dependency files (`dependencies.h`) to track includes.
|
||||
|
||||
- **Amalgamator File**: A file that orchestrates the inclusion of amalgamated files. Examples: `arm64.h`, `arm64/implementation.h`, `generic/amalgamated.h`. These are not themselves amalgamated but control conditional inclusions.
|
||||
|
||||
- **Free Dependency File**: A top-level header that is always included unconditionally. These do not have dependency files and represent the public API (e.g., main headers).
|
||||
|
||||
- **Implementation-Specific File**: A file tied to a specific CPU architecture or instruction set (e.g., `arm64/`, `haswell/`). These must be amalgamated.
|
||||
|
||||
- **Generic File**: A shared file (under `generic/` or `simdjson/generic/`) that contains common code included once per implementation.
|
||||
|
||||
- **Builtin File**: Special files under `simdjson/builtin/` that handle the builtin implementation, a fallback/default implementation used when no optimized implementation is available.
|
||||
|
||||
- **Conditional Include Block**: A section wrapped in `#ifndef SIMDJSON_CONDITIONAL_INCLUDE` for editor-only or implementation-specific content.
|
||||
|
||||
The script `singleheader/amalgation_helper.py` will generate an HTML report which you can use to visualize the status of each file.
|
||||
|
||||
simdjson's source structure, from the top level, looks like this:
|
||||
|
||||
* **CMakeLists.txt:** The main build system.
|
||||
@@ -133,6 +151,12 @@ simdjson's source structure, from the top level, looks like this:
|
||||
* simdjson/generic/ondemand/*.h: individual On-Demand classes, generically written.
|
||||
* simdjson/generic/ondemand/dependencies.h: dependencies on common, non-implementation-specific simdjson classes. This will be included before including amalgamated.h.
|
||||
* simdjson/generic/ondemand/amalgamated.h: all generic ondemand classes for an implementation.
|
||||
* simdjson/builder.h: the `simdjson::builder` namespace. Includes all public builder classes.
|
||||
* simdjson/builtin/builder.h: the `simdjson::builtin::builder` namespace.
|
||||
* simdjson/arm64|fallback|haswell|icelake|ppc64|westmere/builder.h: the `simdjson::<implementation>::builder` namespace. Builder compiled for the specific implementation.
|
||||
* simdjson/generic/builder/*.h: individual Builder classes, generically written.
|
||||
* simdjson/generic/builder/dependencies.h: dependencies on common, non-implementation-specific simdjson classes. This will be included before including amalgamated.h.
|
||||
* simdjson/generic/builder/amalgamated.h: all generic builder classes for an implementation.
|
||||
* **src:** The source files for non-inlined functionality (e.g. the architecture-specific parser
|
||||
implementations).
|
||||
* simdjson.cpp: A "main source" that includes all implementation files from src/. This is
|
||||
@@ -147,6 +171,7 @@ Other important files and directories:
|
||||
* **.github/workflows:** Definitions for GitHub Actions (CI).
|
||||
* **singleheader:** Contains generated `simdjson.h` and `simdjson.cpp` that we release. The files `singleheader/simdjson.h` and `singleheader/simdjson.cpp` should never be edited by hand.
|
||||
* **singleheader/amalgamate.py:** Generates `singleheader/simdjson.h` and `singleheader/simdjson.cpp` for release (python script). If you add a new implementation (e.g., rvv), you need to edit this file (IMPLEMENTATIONS).
|
||||
* **singleheader/amalgation_helper.py:** Generates and `amalgamation_report.html` that helps you understand the status of each file.
|
||||
* **benchmark:** This is where we do benchmarking. Benchmarking is core to every change we make; the
|
||||
cardinal rule is don't regress performance without knowing exactly why, and what you're trading
|
||||
for it. Many of our benchmarks are microbenchmarks. We are effectively doing controlled scientific experiments for the purpose of understanding what affects our performance. So we simplify as much as possible. We try to avoid irrelevant factors such as page faults, interrupts, unnecessary system calls. We recommend checking the performance as follows:
|
||||
|
||||
@@ -6,7 +6,8 @@
|
||||
simdjson : Parsing gigabytes of JSON per second
|
||||
===============================================
|
||||
|
||||
<img src="images/logo.png" width="10%" style="float: right">
|
||||
<img src="images/official_logo/logo_noir/SVG/logo_simdjson_noir.svg" width="40%" style="float: right">
|
||||
|
||||
JSON is everywhere on the Internet. Servers spend a *lot* of time parsing it. We need a fresh
|
||||
approach. The simdjson library uses commonly available SIMD instructions and microparallel algorithms
|
||||
to parse JSON 4x faster than RapidJSON and 25x faster than JSON for Modern C++.
|
||||
@@ -67,6 +68,9 @@ Real-world usage
|
||||
|
||||
If you are planning to use simdjson in a product, please work from one of our releases.
|
||||
|
||||
|
||||
|
||||
|
||||
Quick Start
|
||||
-----------
|
||||
|
||||
@@ -83,7 +87,7 @@ The simdjson library is easily consumable with a single .h and .cpp file.
|
||||
```
|
||||
2. Create `quickstart.cpp`:
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
#include <iostream>
|
||||
#include "simdjson.h"
|
||||
using namespace simdjson;
|
||||
@@ -113,6 +117,8 @@ Usage documentation is available:
|
||||
* [Implementation Selection](doc/implementation-selection.md) describes runtime CPU detection and
|
||||
how you can work with it.
|
||||
* [API](https://simdjson.github.io/simdjson/) contains the automatically generated API documentation.
|
||||
* [Compile-Time Parsing](doc/compile_time.md) presents our compile-time parsing function (C++26 only).
|
||||
|
||||
|
||||
Godbolt
|
||||
-------------
|
||||
@@ -207,6 +213,21 @@ For the video inclined, <br />
|
||||
[](http://www.youtube.com/watch?v=wlvKAT7SZIQ)<br />
|
||||
(It was the best voted talk, we're kinda proud of it.)
|
||||
|
||||
Citing this work
|
||||
-----------------
|
||||
|
||||
If you use simdjson in published research, please cite the software library. A suitable BibTeX entry is:
|
||||
|
||||
```bibtex
|
||||
@misc{simdjson,
|
||||
title={{The simdjson library: Parsing Gigabytes of JSON per Second}},
|
||||
author={Daniel Lemire and Geoff Langdale and John Keiser and Paul Dreik and Francisco Thiesen and others},
|
||||
year={2019},
|
||||
howpublished={Software library},
|
||||
note={https://github.com/simdjson/simdjson}
|
||||
}
|
||||
```
|
||||
|
||||
Funding
|
||||
-------
|
||||
|
||||
@@ -227,6 +248,13 @@ Contributing to simdjson
|
||||
Head over to [CONTRIBUTING.md](CONTRIBUTING.md) for information on contributing to simdjson, and
|
||||
[HACKING.md](HACKING.md) for information on source, building, and architecture/design.
|
||||
|
||||
|
||||
Stars
|
||||
------
|
||||
|
||||
[](https://www.star-history.com/#simdjson/simdjson&Date)
|
||||
|
||||
|
||||
License
|
||||
-------
|
||||
|
||||
|
||||
@@ -0,0 +1,46 @@
|
||||
# Accessor Performance Benchmarks (C++26)
|
||||
|
||||
These benchmarks compare the performance of runtime vs compile-time JSON accessors.
|
||||
For the comparison to be meaningful, you must build simdjson with support for
|
||||
C++26 reflexion. See the `p2996` repository in the main project directory.
|
||||
|
||||
## Files
|
||||
|
||||
- `accessor_benchmark.h` - Common benchmark framework and test data
|
||||
- `runtime_accessors.h` - Runtime `at_path()` benchmarks
|
||||
- `compile_time_accessors.h` - Compile-time `at_path_compiled()` benchmarks (requires C++26 reflection)
|
||||
|
||||
## Benchmarks
|
||||
|
||||
Each benchmark measures parsing + single field access:
|
||||
|
||||
1. **accessor_simple** - Simple field: `.name`
|
||||
2. **accessor_nested** - Nested field: `.address.city`
|
||||
3. **accessor_deep** - Deep nested field: `.address.coordinates.lat`
|
||||
|
||||
## Building (Linux/macOS)
|
||||
|
||||
```bash
|
||||
cmake -B build -D SIMDJSON_STATIC_REFLECTION=ON -DSIMDJSON_DEVELOPER_MODE=ON
|
||||
cmake --build build --target=bench_ondemand
|
||||
```
|
||||
|
||||
The `SIMDJSON_STATIC_REFLECTION` will be made unnecessary once mainstream compilers
|
||||
begin supporting C++26 sufficiently well.
|
||||
|
||||
## Running (Linux/macOS)
|
||||
|
||||
|
||||
```bash
|
||||
# Run all accessor benchmarks
|
||||
./build/bench_ondemand --benchmark_filter="accessor"
|
||||
```
|
||||
|
||||
## Results
|
||||
|
||||
We find that compile-time accessors show performance improvements that scale with path depth:
|
||||
- Simple fields: ~1.2x faster
|
||||
- Nested fields: ~1.5x faster
|
||||
- Deep nested fields: ~1.8x faster
|
||||
|
||||
The speedup comes from eliminating runtime path parsing and conversion overhead.
|
||||
@@ -0,0 +1,132 @@
|
||||
#pragma once
|
||||
|
||||
#include "json_benchmark/file_runner.h"
|
||||
#include <string>
|
||||
|
||||
namespace accessor_performance {
|
||||
|
||||
using namespace json_benchmark;
|
||||
|
||||
// Test JSON for accessor benchmarks
|
||||
static const char* TEST_JSON = R"({
|
||||
"name": "Alice",
|
||||
"age": 30,
|
||||
"email": "alice@example.com",
|
||||
"address": {
|
||||
"street": "123 Main St",
|
||||
"city": "Boston",
|
||||
"state": "MA",
|
||||
"zip": 12345,
|
||||
"coordinates": {
|
||||
"lat": 42.3601,
|
||||
"lon": -71.0589
|
||||
}
|
||||
},
|
||||
"scores": [95, 87, 92, 88, 91],
|
||||
"preferences": {
|
||||
"theme": "dark",
|
||||
"notifications": {
|
||||
"email": true,
|
||||
"push": false,
|
||||
"sms": true
|
||||
}
|
||||
}
|
||||
})";
|
||||
|
||||
// Struct definitions for compile-time validation
|
||||
#if SIMDJSON_STATIC_REFLECTION
|
||||
struct Coordinates {
|
||||
double lat;
|
||||
double lon;
|
||||
};
|
||||
|
||||
struct Address {
|
||||
std::string street;
|
||||
std::string city;
|
||||
std::string state;
|
||||
int64_t zip;
|
||||
Coordinates coordinates;
|
||||
};
|
||||
|
||||
struct Notifications {
|
||||
bool email;
|
||||
bool push;
|
||||
bool sms;
|
||||
};
|
||||
|
||||
struct Preferences {
|
||||
std::string theme;
|
||||
Notifications notifications;
|
||||
};
|
||||
|
||||
struct TestData {
|
||||
std::string name;
|
||||
int64_t age;
|
||||
std::string email;
|
||||
Address address;
|
||||
std::vector<int64_t> scores;
|
||||
Preferences preferences;
|
||||
};
|
||||
#endif // SIMDJSON_STATIC_REFLECTION
|
||||
|
||||
// Single-access benchmark runner: measures ONE field access per iteration
|
||||
template<typename I>
|
||||
struct single_access_runner : public file_runner<I> {
|
||||
std::string result_string;
|
||||
int64_t result_int{};
|
||||
double result_double{};
|
||||
bool result_bool{};
|
||||
|
||||
bool setup(benchmark::State &state) {
|
||||
this->json = simdjson::padded_string(TEST_JSON, strlen(TEST_JSON));
|
||||
state.SetBytesProcessed(int64_t(state.iterations()) * int64_t(this->json.size()));
|
||||
return true;
|
||||
}
|
||||
|
||||
bool before_run(benchmark::State &state) {
|
||||
if (!file_runner<I>::before_run(state)) { return false; }
|
||||
result_string.clear();
|
||||
result_int = 0;
|
||||
result_double = 0.0;
|
||||
result_bool = false;
|
||||
return true;
|
||||
}
|
||||
|
||||
bool run(benchmark::State &) {
|
||||
return this->implementation.run(this->json, result_string, result_int, result_double, result_bool);
|
||||
}
|
||||
|
||||
template<typename R>
|
||||
bool diff(benchmark::State &state, single_access_runner<R> &reference) {
|
||||
if (result_string != reference.result_string ||
|
||||
result_int != reference.result_int ||
|
||||
result_double != reference.result_double ||
|
||||
result_bool != reference.result_bool) {
|
||||
std::cerr << "Accessor benchmark results differ!" << std::endl;
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
size_t items_per_iteration() {
|
||||
return 1;
|
||||
}
|
||||
};
|
||||
|
||||
// Benchmark template definitions
|
||||
struct runtime_at_path_simple;
|
||||
template<typename I> simdjson_inline static void accessor_simple(benchmark::State &state) {
|
||||
run_json_benchmark<single_access_runner<I>, single_access_runner<runtime_at_path_simple>>(state);
|
||||
}
|
||||
|
||||
struct runtime_at_path_nested;
|
||||
template<typename I> simdjson_inline static void accessor_nested(benchmark::State &state) {
|
||||
run_json_benchmark<single_access_runner<I>, single_access_runner<runtime_at_path_nested>>(state);
|
||||
}
|
||||
|
||||
struct runtime_at_path_deep;
|
||||
template<typename I> simdjson_inline static void accessor_deep(benchmark::State &state) {
|
||||
run_json_benchmark<single_access_runner<I>, single_access_runner<runtime_at_path_deep>>(state);
|
||||
}
|
||||
|
||||
} // namespace accessor_performance
|
||||
@@ -0,0 +1,56 @@
|
||||
#pragma once
|
||||
|
||||
#if SIMDJSON_EXCEPTIONS && SIMDJSON_STATIC_REFLECTION
|
||||
|
||||
#include "accessor_benchmark.h"
|
||||
|
||||
namespace accessor_performance {
|
||||
|
||||
using namespace simdjson;
|
||||
|
||||
struct compile_time_at_path_simple {
|
||||
ondemand::parser parser{};
|
||||
|
||||
bool run(simdjson::padded_string &json, std::string &result_str, int64_t&, double&, bool&) {
|
||||
auto doc = parser.iterate(json);
|
||||
std::string_view name;
|
||||
auto r = ondemand::json_path::at_path_compiled<TestData, ".name">(doc);
|
||||
if (r.get(name) != SUCCESS) return false;
|
||||
result_str = name;
|
||||
return true;
|
||||
}
|
||||
};
|
||||
|
||||
struct compile_time_at_path_nested {
|
||||
ondemand::parser parser{};
|
||||
|
||||
bool run(simdjson::padded_string &json, std::string &result_str, int64_t&, double&, bool&) {
|
||||
auto doc = parser.iterate(json);
|
||||
std::string_view city;
|
||||
auto r = ondemand::json_path::at_path_compiled<TestData, ".address.city">(doc);
|
||||
if (r.get(city) != SUCCESS) return false;
|
||||
result_str = city;
|
||||
return true;
|
||||
}
|
||||
};
|
||||
|
||||
struct compile_time_at_path_deep {
|
||||
ondemand::parser parser{};
|
||||
|
||||
bool run(simdjson::padded_string &json, std::string&, int64_t&, double &result_dbl, bool&) {
|
||||
auto doc = parser.iterate(json);
|
||||
double lat;
|
||||
auto r = ondemand::json_path::at_path_compiled<TestData, ".address.coordinates.lat">(doc);
|
||||
if (r.get(lat) != SUCCESS) return false;
|
||||
result_dbl = lat;
|
||||
return true;
|
||||
}
|
||||
};
|
||||
|
||||
BENCHMARK_TEMPLATE(accessor_simple, compile_time_at_path_simple)->UseManualTime();
|
||||
BENCHMARK_TEMPLATE(accessor_nested, compile_time_at_path_nested)->UseManualTime();
|
||||
BENCHMARK_TEMPLATE(accessor_deep, compile_time_at_path_deep)->UseManualTime();
|
||||
|
||||
} // namespace accessor_performance
|
||||
|
||||
#endif // SIMDJSON_EXCEPTIONS && SIMDJSON_STATIC_REFLECTION
|
||||
@@ -0,0 +1,53 @@
|
||||
#pragma once
|
||||
|
||||
#if SIMDJSON_EXCEPTIONS
|
||||
|
||||
#include "accessor_benchmark.h"
|
||||
|
||||
namespace accessor_performance {
|
||||
|
||||
using namespace simdjson;
|
||||
|
||||
struct runtime_at_path_simple {
|
||||
ondemand::parser parser{};
|
||||
|
||||
bool run(simdjson::padded_string &json, std::string &result_str, int64_t&, double&, bool&) {
|
||||
auto doc = parser.iterate(json);
|
||||
std::string_view name;
|
||||
if (doc.at_path(".name").get(name) != SUCCESS) return false;
|
||||
result_str = name;
|
||||
return true;
|
||||
}
|
||||
};
|
||||
|
||||
struct runtime_at_path_nested {
|
||||
ondemand::parser parser{};
|
||||
|
||||
bool run(simdjson::padded_string &json, std::string &result_str, int64_t&, double&, bool&) {
|
||||
auto doc = parser.iterate(json);
|
||||
std::string_view city;
|
||||
if (doc.at_path(".address.city").get(city) != SUCCESS) return false;
|
||||
result_str = city;
|
||||
return true;
|
||||
}
|
||||
};
|
||||
|
||||
struct runtime_at_path_deep {
|
||||
ondemand::parser parser{};
|
||||
|
||||
bool run(simdjson::padded_string &json, std::string&, int64_t&, double &result_dbl, bool&) {
|
||||
auto doc = parser.iterate(json);
|
||||
double lat;
|
||||
if (doc.at_path(".address.coordinates.lat").get(lat) != SUCCESS) return false;
|
||||
result_dbl = lat;
|
||||
return true;
|
||||
}
|
||||
};
|
||||
|
||||
BENCHMARK_TEMPLATE(accessor_simple, runtime_at_path_simple)->UseManualTime();
|
||||
BENCHMARK_TEMPLATE(accessor_nested, runtime_at_path_nested)->UseManualTime();
|
||||
BENCHMARK_TEMPLATE(accessor_deep, runtime_at_path_deep)->UseManualTime();
|
||||
|
||||
} // namespace accessor_performance
|
||||
|
||||
#endif // SIMDJSON_EXCEPTIONS
|
||||
@@ -245,7 +245,7 @@ static u32 (*kpc_get_counter_count)(u32 classes);
|
||||
|
||||
/// Get counter accumulations.
|
||||
/// If `all_cpus` is true, the buffer count should not smaller than
|
||||
/// (cpu_count * counter_count). Otherwize, the buffer count should not smaller
|
||||
/// (cpu_count * counter_count). Otherwise, the buffer count should not smaller
|
||||
/// than (counter_count).
|
||||
/// @see kpc_get_counter_count(), kpc_cpu_count().
|
||||
/// @param all_cpus true for all CPUs, false for current cpu.
|
||||
@@ -374,7 +374,7 @@ static int kperf_lightweight_pet_set(u32 enabled) {
|
||||
// These functions do not require root privileges.
|
||||
// -----------------------------------------------------------------------------
|
||||
|
||||
// KPEP CPU archtecture constants.
|
||||
// KPEP CPU architecture constants.
|
||||
#define KPEP_ARCH_I386 0
|
||||
#define KPEP_ARCH_X86_64 1
|
||||
#define KPEP_ARCH_ARM 2
|
||||
@@ -414,7 +414,7 @@ typedef struct kpep_db {
|
||||
usize fixed_counter_count;
|
||||
usize config_counter_count;
|
||||
usize power_counter_count;
|
||||
u32 archtecture; ///< see `KPEP CPU archtecture constants` above.
|
||||
u32 architecture; ///< see `KPEP CPU architecture constants` above.
|
||||
u32 fixed_counter_bits;
|
||||
u32 config_counter_bits;
|
||||
u32 power_counter_bits;
|
||||
|
||||
@@ -148,4 +148,9 @@ SIMDJSON_POP_DISABLE_WARNINGS
|
||||
#include "large_amazon_cellphones/simdjson_dom.h"
|
||||
#include "large_amazon_cellphones/simdjson_ondemand.h"
|
||||
|
||||
#include "accessor_performance/runtime_accessors.h"
|
||||
#if SIMDJSON_STATIC_REFLECTION
|
||||
#include "accessor_performance/compile_time_accessors.h"
|
||||
#endif
|
||||
|
||||
BENCHMARK_MAIN();
|
||||
|
||||
@@ -44,7 +44,10 @@ struct yyjson_base {
|
||||
|
||||
struct yyjson : yyjson_base {
|
||||
bool run(simdjson::padded_string &json, std::vector<uint64_t> &result) {
|
||||
return yyjson_base::run(yyjson_read(json.data(), json.size(), 0), result);
|
||||
yyjson_doc *doc = yyjson_read(json.data(), json.size(), 0);
|
||||
bool b = yyjson_base::run(doc, result);
|
||||
yyjson_doc_free(doc);
|
||||
return b;
|
||||
}
|
||||
};
|
||||
BENCHMARK_TEMPLATE(distinct_user_id, yyjson)->UseManualTime();
|
||||
@@ -52,11 +55,15 @@ BENCHMARK_TEMPLATE(distinct_user_id, yyjson)->UseManualTime();
|
||||
#if SIMDJSON_COMPETITION_ONDEMAND_INSITU
|
||||
struct yyjson_insitu : yyjson_base {
|
||||
bool run(simdjson::padded_string &json, std::vector<uint64_t> &result) {
|
||||
return yyjson_base::run(yyjson_read_opts(json.data(), json.size(), YYJSON_READ_INSITU, 0, 0), result);
|
||||
yyjson_doc *doc = yyjson_read_opts(json.data(), json.size(), YYJSON_READ_INSITU, 0, 0);
|
||||
bool b = yyjson_base::run(doc, result);
|
||||
yyjson_doc_free(doc);
|
||||
return b;
|
||||
}
|
||||
};
|
||||
BENCHMARK_TEMPLATE(distinct_user_id, yyjson_insitu)->UseManualTime();
|
||||
#endif // SIMDJSON_COMPETITION_ONDEMAND_INSITU
|
||||
|
||||
} // namespace distinct_user_id
|
||||
|
||||
#endif // SIMDJSON_COMPETITION_YYJSON
|
||||
|
||||
@@ -36,18 +36,26 @@ struct yyjson_base {
|
||||
|
||||
struct yyjson : yyjson_base {
|
||||
bool run(simdjson::padded_string &json, uint64_t find_id, std::string_view &result) {
|
||||
return yyjson_base::run(yyjson_read(json.data(), json.size(), 0), find_id, result);
|
||||
yyjson_doc *doc = yyjson_read(json.data(), json.size(), 0);
|
||||
bool b = yyjson_base::run(doc, find_id, result);
|
||||
yyjson_doc_free(doc);
|
||||
return b;
|
||||
}
|
||||
};
|
||||
BENCHMARK_TEMPLATE(find_tweet, yyjson)->UseManualTime();
|
||||
|
||||
#if SIMDJSON_COMPETITION_ONDEMAND_INSITU
|
||||
struct yyjson_insitu : yyjson_base {
|
||||
bool run(simdjson::padded_string &json, uint64_t find_id, std::string_view &result) {
|
||||
return yyjson_base::run(yyjson_read_opts(json.data(), json.size(), YYJSON_READ_INSITU, 0, 0), find_id, result);
|
||||
yyjson_doc *doc = yyjson_read_opts(json.data(), json.size(), YYJSON_READ_INSITU, 0, 0);
|
||||
bool b = yyjson_base::run(doc, find_id, result);
|
||||
yyjson_doc_free(doc);
|
||||
return b;
|
||||
}
|
||||
};
|
||||
BENCHMARK_TEMPLATE(find_tweet, yyjson_insitu)->UseManualTime();
|
||||
#endif // SIMDJSON_COMPETITION_ONDEMAND_INSITU
|
||||
|
||||
} // namespace find_tweet
|
||||
|
||||
#endif // SIMDJSON_COMPETITION_YYJSON
|
||||
|
||||
@@ -100,6 +100,7 @@ struct yyjson : yyjson2msgpack {
|
||||
std::string_view &result) {
|
||||
yyjson_doc *doc = yyjson_read(json.data(), json.size(), 0);
|
||||
result = to_msgpack(doc, reinterpret_cast<uint8_t*>(buffer));
|
||||
yyjson_doc_free(doc);
|
||||
return true;
|
||||
}
|
||||
};
|
||||
@@ -113,6 +114,7 @@ struct yyjson_insitu : yyjson2msgpack {
|
||||
yyjson_doc *doc =
|
||||
yyjson_read_opts(json.data(), json.size(), YYJSON_READ_INSITU, 0, 0);
|
||||
result = to_msgpack(doc, reinterpret_cast<uint8_t*>(buffer));
|
||||
yyjson_doc_free(doc);
|
||||
return true;
|
||||
}
|
||||
};
|
||||
@@ -120,4 +122,4 @@ BENCHMARK_TEMPLATE(json2msgpack, yyjson_insitu)->UseManualTime();
|
||||
#endif // SIMDJSON_COMPETITION_ONDEMAND_INSITU
|
||||
} // namespace json2msgpack
|
||||
|
||||
#endif // SIMDJSON_COMPETITION_YYJSON
|
||||
#endif // SIMDJSON_COMPETITION_YYJSON
|
||||
|
||||
@@ -49,18 +49,26 @@ struct yyjson_base {
|
||||
|
||||
struct yyjson : yyjson_base {
|
||||
bool run(simdjson::padded_string &json, std::vector<point> &result) {
|
||||
return yyjson_base::run(yyjson_read(json.data(), json.size(), 0), result);
|
||||
yyjson_doc *doc = yyjson_read(json.data(), json.size(), 0);
|
||||
bool b = yyjson_base::run(doc, result);
|
||||
yyjson_doc_free(doc);
|
||||
return b;
|
||||
}
|
||||
};
|
||||
BENCHMARK_TEMPLATE(kostya, yyjson)->UseManualTime();
|
||||
|
||||
#if SIMDJSON_COMPETITION_ONDEMAND_INSITU
|
||||
struct yyjson_insitu : yyjson_base {
|
||||
bool run(simdjson::padded_string &json, std::vector<point> &result) {
|
||||
return yyjson_base::run(yyjson_read_opts(json.data(), json.size(), YYJSON_READ_INSITU, 0, 0), result);
|
||||
yyjson_doc *doc = yyjson_read_opts(json.data(), json.size(), YYJSON_READ_INSITU, 0, 0);
|
||||
bool b = yyjson_base::run(doc, result);
|
||||
yyjson_doc_free(doc);
|
||||
return b;
|
||||
}
|
||||
};
|
||||
BENCHMARK_TEMPLATE(kostya, yyjson_insitu)->UseManualTime();
|
||||
#endif // SIMDJSON_COMPETITION_ONDEMAND_INSITU
|
||||
|
||||
} // namespace kostya
|
||||
|
||||
#endif // SIMDJSON_COMPETITION_YYJSON
|
||||
|
||||
@@ -47,18 +47,26 @@ struct yyjson_base {
|
||||
|
||||
struct yyjson : yyjson_base {
|
||||
bool run(simdjson::padded_string &json, std::vector<point> &result) {
|
||||
return yyjson_base::run(yyjson_read(json.data(), json.size(), 0), result);
|
||||
yyjson_doc *doc = yyjson_read(json.data(), json.size(), 0);
|
||||
bool b = yyjson_base::run(doc, result);
|
||||
yyjson_doc_free(doc);
|
||||
return b;
|
||||
}
|
||||
};
|
||||
BENCHMARK_TEMPLATE(large_random, yyjson)->UseManualTime();
|
||||
|
||||
#if SIMDJSON_COMPETITION_ONDEMAND_INSITU
|
||||
struct yyjson_insitu : yyjson_base {
|
||||
bool run(simdjson::padded_string &json, std::vector<point> &result) {
|
||||
return yyjson_base::run(yyjson_read_opts(json.data(), json.size(), YYJSON_READ_INSITU, 0, 0), result);
|
||||
yyjson_doc *doc = yyjson_read_opts(json.data(), json.size(), YYJSON_READ_INSITU, 0, 0);
|
||||
bool b = yyjson_base::run(doc, result);
|
||||
yyjson_doc_free(doc);
|
||||
return b;
|
||||
}
|
||||
};
|
||||
BENCHMARK_TEMPLATE(large_random, yyjson_insitu)->UseManualTime();
|
||||
#endif // SIMDJSON_COMPETITION_ONDEMAND_INSITU
|
||||
|
||||
} // namespace large_random
|
||||
|
||||
#endif // SIMDJSON_COMPETITION_YYJSON
|
||||
|
||||
@@ -62,19 +62,26 @@ struct yyjson_base {
|
||||
|
||||
struct yyjson : yyjson_base {
|
||||
bool run(simdjson::padded_string &json, std::vector<tweet<std::string_view>> &result) {
|
||||
return yyjson_base::run(yyjson_read(json.data(), json.size(), 0), result);
|
||||
yyjson_doc *doc = yyjson_read(json.data(), json.size(), 0);
|
||||
bool b = yyjson_base::run(doc, result);
|
||||
yyjson_doc_free(doc);
|
||||
return b;
|
||||
}
|
||||
};
|
||||
BENCHMARK_TEMPLATE(partial_tweets, yyjson)->UseManualTime();
|
||||
|
||||
#if SIMDJSON_COMPETITION_ONDEMAND_INSITU
|
||||
struct yyjson_insitu : yyjson_base {
|
||||
bool run(simdjson::padded_string &json, std::vector<tweet<std::string_view>> &result) {
|
||||
return yyjson_base::run(yyjson_read_opts(json.data(), json.size(), YYJSON_READ_INSITU, 0, 0), result);
|
||||
yyjson_doc *doc = yyjson_read_opts(json.data(), json.size(), YYJSON_READ_INSITU, 0, 0);
|
||||
bool b = yyjson_base::run(doc, result);
|
||||
yyjson_doc_free(doc);
|
||||
return b;
|
||||
}
|
||||
};
|
||||
BENCHMARK_TEMPLATE(partial_tweets, yyjson_insitu)->UseManualTime();
|
||||
#endif // SIMDJSON_COMPETITION_ONDEMAND_INSITU
|
||||
|
||||
} // namespace partial_tweets
|
||||
|
||||
#endif // SIMDJSON_COMPETITION_YYJSON
|
||||
|
||||
|
||||
@@ -51,18 +51,26 @@ struct yyjson_base {
|
||||
|
||||
struct yyjson : yyjson_base {
|
||||
bool run(simdjson::padded_string &json, int64_t max_retweet_count, top_tweet_result<StringType> &result) {
|
||||
return yyjson_base::run(yyjson_read(json.data(), json.size(), 0), max_retweet_count, result);
|
||||
yyjson_doc *doc = yyjson_read(json.data(), json.size(), 0);
|
||||
bool b = yyjson_base::run(doc, max_retweet_count, result);
|
||||
yyjson_doc_free(doc);
|
||||
return b;
|
||||
}
|
||||
};
|
||||
BENCHMARK_TEMPLATE(top_tweet, yyjson)->UseManualTime();
|
||||
|
||||
#if SIMDJSON_COMPETITION_ONDEMAND_INSITU
|
||||
struct yyjson_insitu : yyjson_base {
|
||||
bool run(simdjson::padded_string &json, int64_t max_retweet_count, top_tweet_result<StringType> &result) {
|
||||
return yyjson_base::run(yyjson_read_opts(json.data(), json.size(), YYJSON_READ_INSITU, 0, 0), max_retweet_count, result);
|
||||
yyjson_doc *doc = yyjson_read_opts(json.data(), json.size(), YYJSON_READ_INSITU, 0, 0);
|
||||
bool b = yyjson_base::run(doc, max_retweet_count, result);
|
||||
yyjson_doc_free(doc);
|
||||
return b;
|
||||
}
|
||||
};
|
||||
BENCHMARK_TEMPLATE(top_tweet, yyjson_insitu)->UseManualTime();
|
||||
#endif // SIMDJSON_COMPETITION_ONDEMAND_INSITU
|
||||
|
||||
} // namespace top_tweet
|
||||
|
||||
#endif // SIMDJSON_COMPETITION_YYJSON
|
||||
|
||||
@@ -17,8 +17,9 @@ editing CMAKE_CXX_FLAGS")
|
||||
# /EHc used in conjection with /EHs indicates that extern "C" functions
|
||||
# never throw (terminate-on-throw)
|
||||
# Here, we disable both with the - argument negation operator
|
||||
string(REPLACE "/EHsc" "/EHs-c-" CMAKE_CXX_FLAGS ${CMAKE_CXX_FLAGS})
|
||||
|
||||
if(CMAKE_CXX_FLAGS)
|
||||
string(REPLACE "/EHsc" "/EHs-c-" CMAKE_CXX_FLAGS ${CMAKE_CXX_FLAGS})
|
||||
endif()
|
||||
# Because we cannot change the flag above on an individual target (yet), the
|
||||
# definition below must similarly be added globally
|
||||
add_definitions(-D_HAS_EXCEPTIONS=0)
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
The Basics
|
||||
==========
|
||||
|
||||
|
||||
An overview of what you need to know to use simdjson to parse JSON documents, with examples.
|
||||
[Our documentation regarding the generation (serialization) of JSON documents is in a
|
||||
separate document](https://github.com/simdjson/simdjson/blob/master/doc/builder.md).
|
||||
@@ -21,11 +22,13 @@ separate document](https://github.com/simdjson/simdjson/blob/master/doc/builder.
|
||||
* [1. Specialize `simdjson::ondemand::value::get` to get custom types (pre-C++20)](#1-specialize-simdjsonondemandvalueget-to-get-custom-types-pre-c20)
|
||||
* [2. Use `tag_invoke` for custom types (C++20)](#2-use-tag_invoke-for-custom-types-c20)
|
||||
* [3. Using static reflection (C++26)](#3-using-static-reflection-c26)
|
||||
+ [Special cases](#special-cases)
|
||||
* [The simdjson::from shortcut (experimental, C++20)](#the-simdjsonfrom-shortcut-experimental-c20)
|
||||
- [Minifying JSON strings without parsing](#minifying-json-strings-without-parsing)
|
||||
- [UTF-8 validation (alone)](#utf-8-validation-alone)
|
||||
- [JSON Pointer](#json-pointer)
|
||||
- [JSONPath](#jsonpath)
|
||||
- [Compile-Time JSONPath and JSON Pointer (C++26 Reflection)](#compile-time-jsonpath-and-json-pointer-c26-reflection)
|
||||
- [Error handling](#error-handling)
|
||||
* [Error handling examples without exceptions](#error-handling-examples-without-exceptions)
|
||||
* [Disabling exceptions](#disabling-exceptions)
|
||||
@@ -66,7 +69,7 @@ Including simdjson
|
||||
To include simdjson, copy [simdjson.h](/singleheader/simdjson.h) and [simdjson.cpp](/singleheader/simdjson.cpp)
|
||||
into your project. Then include it in your project with:
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
#include "simdjson.h"
|
||||
using namespace simdjson; // optional
|
||||
```
|
||||
@@ -166,25 +169,30 @@ For efficiency reasons, simdjson requires a string with a few bytes (`simdjson::
|
||||
at the end, these bytes may be read but their content does not affect the parsing. In practice,
|
||||
it means that the JSON inputs should be stored in a memory region with `simdjson::SIMDJSON_PADDING`
|
||||
extra bytes at the end. You do not have to set these bytes to specific values though you may
|
||||
want to if you want to avoid runtime warnings with some sanitizers. Advanced users may want to
|
||||
read the section Free Padding in [our performance notes](performance.md).
|
||||
want to if you want to avoid runtime warnings with some sanitizers. We expect the user
|
||||
of the library to load the data (from disk or from the network) into a padded buffer. To make
|
||||
this easy, we provide the `padded_string::load` function which loads files from disk in a padded buffer.
|
||||
[You can similarly fetch a file from a URL to a padded string](https://github.com/simdjson/curltostring) using our `simdjson::padded_string_builder`. Advanced users may want to read the section Free Padding in [our performance notes](performance.md).
|
||||
|
||||
The simdjson library offers a tree-like [API](https://en.wikipedia.org/wiki/API), which you can
|
||||
access by creating a `ondemand::parser` and calling the `iterate()` method. The iterate method
|
||||
quickly indexes the input string and may detect some errors. The following example illustrates
|
||||
how to get started with an input JSON file (`"twitter.json"`):
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
ondemand::parser parser;
|
||||
auto json = padded_string::load("twitter.json"); // load JSON file 'twitter.json'.
|
||||
ondemand::document doc = parser.iterate(json); // position a pointer at the beginning of the JSON data
|
||||
```
|
||||
|
||||
(Windows users compiling with C++17 or better may use `wchar_t` strings to support non-ASCII
|
||||
filenames: `padded_string::load(L"twitter.json")`.)
|
||||
|
||||
If you prefer not to create your own `ondemand::parser` instance, you can access
|
||||
a thread-local version by calling `ondemand::parser.get_parser()`.
|
||||
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
ondemand::document doc = ondemand::parser.get_parser().iterate(json);
|
||||
```
|
||||
|
||||
@@ -194,7 +202,7 @@ document per thread at any one time.
|
||||
|
||||
You can also create a padded string---and call `iterate()`:
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
ondemand::parser parser;
|
||||
auto json = "[1,2,3]"_padded; // The _padded suffix creates a simdjson::padded_string instance
|
||||
ondemand::document doc = parser.iterate(json); // parse a string
|
||||
@@ -202,7 +210,7 @@ ondemand::document doc = parser.iterate(json); // parse a string
|
||||
|
||||
If you have a buffer of your own with enough padding already (SIMDJSON_PADDING extra bytes allocated), you can use `padded_string_view` to pass it in:
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
ondemand::parser parser;
|
||||
char json[3+SIMDJSON_PADDING];
|
||||
strcpy(json, "[1]");
|
||||
@@ -214,14 +222,14 @@ reference is non-const, it will allocate padding as needed.
|
||||
|
||||
You can copy your data directly on a `simdjson::padded_string` as follows:
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
const char * data = "my data"; // 7 bytes
|
||||
simdjson::padded_string my_padded_data(data, 7); // copies to a padded buffer
|
||||
```
|
||||
|
||||
Or as follows...
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
std::string data = "my data";
|
||||
simdjson::padded_string my_padded_data(data); // copies to a padded buffer
|
||||
```
|
||||
@@ -229,7 +237,7 @@ simdjson::padded_string my_padded_data(data); // copies to a padded buffer
|
||||
You can then parse the JSON data from the `simdjson::padded_string` instance:
|
||||
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
ondemand::document doc = parser.iterate(my_padded_data);
|
||||
```
|
||||
|
||||
@@ -241,7 +249,7 @@ container-overflow checks, you may encounter sanitizer warnings.
|
||||
You can safely ignore these warnings. Or you can call `simdjson::pad(std::string&)` to pad the
|
||||
string with `SIMDJSON_PADDING` spaces: this function returns a `simdjson::padding_string_view` which can be be passed to the parser's iterator function:
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
std::string json = "[1]";
|
||||
ondemand::document doc = parser.iterate(simdjson::pad(json));
|
||||
```
|
||||
@@ -348,7 +356,13 @@ the macro `SIMDJSON_DEVELOPMENT_CHECKS` to 1 prior to including
|
||||
the `simdjson.h` header to enable these additional checks: just make sure you remove the
|
||||
definition once your code has been tested. When `SIMDJSON_DEVELOPMENT_CHECKS` is set to 1, the
|
||||
simdjson library runs additional (expensive) tests on your code to help ensure that you are
|
||||
using the library in a safe manner.
|
||||
using the library in a safe manner. We add asserts which may halt your program, helping
|
||||
you find the bad programming pattern.
|
||||
|
||||
When `SIMDJSON_DEVELOPMENT_CHECKS`, some of our data structures contain extra data for
|
||||
tracking explicitly potential programming mistakes. Thus you should not relying on the
|
||||
size (`sizeof`) of our data structures to be constant: they may change depending on the
|
||||
compiler settings.
|
||||
|
||||
Once your code has been tested, you can then run it in
|
||||
Release mode: under Visual Studio, it means having the `_DEBUG` macro undefined, and, for other
|
||||
@@ -431,8 +445,8 @@ support for users who avoid exceptions. See [the simdjson error handling documen
|
||||
|
||||
If you know the type of the value, you can cast it right there, too! `for (double value : array) { ... }`.
|
||||
|
||||
You may also use explicit iterators: `for(auto i = array.begin(); i != array.end(); i++) {}`. You can check that an array is empty with the condition `auto i = array.begin(); if (i == array.end()) {...}`.
|
||||
* **Object Iteration:** You can iterate through an object's fields, as well: `for (auto field : object) { ... }`. You may also use explicit iterators : `for(auto i = object.begin(); i != object.end(); i++) { auto field = *i; .... }`. You can check that an object is empty with the condition `auto i = object.begin(); if (i == object.end()) {...}`.
|
||||
You may also use explicit iterators: `for(auto i = array.begin(); i != array.end(); i++) {}`. You can check that an array is empty with the condition `auto i = array.begin(); if (i == array.end()) {...}`. You should derefence (`*i`) an iterator at most once before incrementing it (`i++`), when compiling in debug mode with development checks, we add asserts to help you identify such a mistake.
|
||||
* **Object Iteration:** You can iterate through an object's fields, as well: `for (auto field : object) { ... }`.
|
||||
- `field.unescaped_key()` will get you the unescaped key string as a `std::string_view` instance. E.g., the JSON string `"\u00e1"` becomes the Unicode string `á`. Optionally, you pass `true` as a parameter to the `unescaped_key` method if you want invalid escape sequences to be replaced by a default replacement character (e.g., `\ud800\ud801\ud811`): otherwise bad escape sequences lead to an immediate error.
|
||||
- `field.escaped_key()` will get you the key string as as a `std::string_view` instance, but unlike `unescaped_key()`, the key is not processed, so no unescaping is done. E.g., the JSON string `"\u00e1"` becomes the Unicode string `\u00e1`. We expect that `escaped_key()` is faster than `field.unescaped_key()`.
|
||||
- `field.value()` will get you the value, which you can then use all these other methods on.
|
||||
@@ -445,8 +459,12 @@ support for users who avoid exceptions. See [the simdjson error handling documen
|
||||
|
||||
When you are iterating through an object, you are advancing through its keys and values. You should not also access the object or other objects. E.g. within a loop over `myobject`, you should not be accessing `myobject`. The following is an anti-pattern: `for(auto value: myobject) {myobject["mykey"]}`.
|
||||
|
||||
We discourage using the object iterators explicitly: `for(auto i = object.begin(); i != object.end(); i++) { auto field = *i; .... }`. In addition to the usual requirement to check against `end()` prior to dereferencing, you must also always dereference the pointer (`*it`) exactly once before you increment it (`it++`). You must also only deference the iterator once (never more than once). When compiling in
|
||||
debug mode with development checks, we add asserts to help check whether you correctly
|
||||
dereferenced the pointer before incrementing it.
|
||||
|
||||
You should never reset an object as you are iterating through it. The following is an anti-pattern: `for(auto value: myobject) {myobject.reset()}`.
|
||||
* **Array Index:** Because it is forward-only, you cannot look up an array element by index by index. Instead,
|
||||
* **Array Index:** Because it is forward-only, you cannot look up an array element by index. Instead,
|
||||
you should iterate through the array and keep an index yourself. Exceptionally, if need a single value
|
||||
out of the array, you may use an array access (e.g., `array[1]`). You should never reset an array as you are iterating through it. The following is an anti-pattern: `for(auto value: myarray) {myarray.reset()}`.
|
||||
* **Field Access:** To get the value of the "foo" field in an object, use `object["foo"]`. This will
|
||||
@@ -474,8 +492,8 @@ support for users who avoid exceptions. See [the simdjson error handling documen
|
||||
> which returns a `std::string_view` instance pointing directly in the document, like `key()`, although,
|
||||
> unlike `key()`, it has to determine the location of the final quote character.
|
||||
>
|
||||
> ```c++
|
||||
> auto json = R"({"k\u0065y": 1})"_padded;
|
||||
> ```cpp
|
||||
> auto json = R"({"k\u0065y": 1})"_padded; // R"( ... )" is a C++ raw string literal.
|
||||
> ondemand::parser parser;
|
||||
> auto doc = parser.iterate(json);
|
||||
> ondemand::object object = doc.get_object();
|
||||
@@ -498,9 +516,9 @@ support for users who avoid exceptions. See [the simdjson error handling documen
|
||||
> This will only look forward, and will fail to find fields in the wrong order: for example, this
|
||||
> will fail:
|
||||
>
|
||||
> ```c++
|
||||
> ```cpp
|
||||
> ondemand::parser parser;
|
||||
> auto json = R"( { "x": 1, "y": 2 } )"_padded;
|
||||
> auto json = R"( { "x": 1, "y": 2 } )"_padded; // R"( ... )" is a C++ raw string literal.
|
||||
> auto doc = parser.iterate(json);
|
||||
> double y = doc.find_field("y"); // The cursor is now after the 2 (at })
|
||||
> double x = doc.find_field("x"); // This fails, because there are no more fields after "y"
|
||||
@@ -508,7 +526,7 @@ support for users who avoid exceptions. See [the simdjson error handling documen
|
||||
>
|
||||
> By contrast, using the default (order-insensitive) lookup succeeds:
|
||||
>
|
||||
> ```c++
|
||||
> ```cpp
|
||||
> ondemand::parser parser;
|
||||
> auto json = R"( { "x": 1, "y": 2 } )"_padded;
|
||||
> auto doc = parser.iterate(json);
|
||||
@@ -516,20 +534,20 @@ support for users who avoid exceptions. See [the simdjson error handling documen
|
||||
> double x = doc["x"]; // Success: [] loops back around to find "x"
|
||||
> ```
|
||||
* **Output to strings:** Given a document, a value, an array or an object in a JSON document, you can output a JSON string version suitable to be parsed again as JSON content: `simdjson::to_json_string(element)`. A call to `to_json_string` consumes fully the element: if you apply it on a document, the internal pointer is advanced to the end of the document. The `simdjson::to_json_string` does not allocate memory. The `to_json_string` function should not be confused with retrieving the value of a string instance which are escaped and represented using a lightweight `std::string_view` instance pointing at an internal string buffer inside the parser instance. To illustrate, the first of the following two code segments will print the unescaped string `"test"` complete with the quote whereas the second one will print the escaped content of the string (without the quotes).
|
||||
> ```C++
|
||||
> ```cpp
|
||||
> // serialize a JSON to an escaped std::string instance so that it can be parsed again as JSON
|
||||
> auto silly_json = R"( { "test": "result" } )"_padded;
|
||||
> ondemand::document doc = parser.iterate(silly_json);
|
||||
> std::cout << simdjson::to_json_string(doc["test"]) << std::endl; // Requires simdjson 1.0 or better
|
||||
>````
|
||||
> ```C++
|
||||
> ```
|
||||
> ```cpp
|
||||
> // retrieves an unescaped string value as a string_view instance
|
||||
> auto silly_json = R"( { "test": "result" } )"_padded;
|
||||
> ondemand::document doc = parser.iterate(silly_json);
|
||||
> std::cout << std::string_view(doc["test"]) << std::endl;
|
||||
>````
|
||||
> ```
|
||||
You can use `to_json_string` to efficiently extract components of a JSON document to reconstruct a new JSON document, as in the following example:
|
||||
> ```C++
|
||||
> ```cpp
|
||||
> auto cars_json = R"( [
|
||||
> { "make": "Toyota", "model": "Camry", "year": 2018, "tire_pressure": [ 40.1, 39.9, 37.7, 40.4 ] },
|
||||
> { "make": "Kia", "model": "Soul", "year": 2012, "tire_pressure": [ 30.1, 31.0, 28.6, 28.7 ] },
|
||||
@@ -556,13 +574,13 @@ support for users who avoid exceptions. See [the simdjson error handling documen
|
||||
> oss << "]";
|
||||
> auto json_string = oss.str();
|
||||
> // json_string == "[[ 40.1, 39.9, 37.7, 40.4 ],[ 30.1, 31.0, 28.6, 28.7 ]]"
|
||||
>````
|
||||
> ```
|
||||
* **Extracting Values (without exceptions):** You can use a variant usage of `get()` with error
|
||||
codes to avoid exceptions. You first declare the variable of the appropriate type (`double`,
|
||||
`uint64_t`, `int64_t`, `bool`, `ondemand::object` and `ondemand::array`) and pass it by reference
|
||||
to `get()` which gives you back an error code: e.g.,
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
auto abstract_json = R"(
|
||||
{ "str" : { "123" : {"abc" : 3.14 } } }
|
||||
)"_padded;
|
||||
@@ -583,7 +601,7 @@ support for users who avoid exceptions. See [the simdjson error handling documen
|
||||
whole array. You should only call `count_elements` as a last resort as it may
|
||||
require scanning the document twice or more. You should never use the `count_elements` as part of an attempt to iterate through the array: use a `for` loop to iterate through arrays. In the spirit of On-Demand, the `count_elements` function does not validate the values in the array: they are validated when they are consumed. You may use it as follows if your document is itself an array:
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
auto cars_json = R"( [ 40.1, 39.9, 37.7, 40.4 ] )"_padded;
|
||||
auto doc = parser.iterate(cars_json);
|
||||
size_t count = doc.count_elements(); // requires simdjson 1.0 or better
|
||||
@@ -611,7 +629,7 @@ support for users who avoid exceptions. See [the simdjson error handling documen
|
||||
whole objects. You should only call `count_fields` as a last resort as it may
|
||||
require scanning the document twice or more. You may use it as follows if your document is itself an object:
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
ondemand::parser parser;
|
||||
auto json = R"( { "test":{ "val1":1, "val2":2 } } )"_padded;
|
||||
auto doc = parser.iterate(json);
|
||||
@@ -642,7 +660,7 @@ support for users who avoid exceptions. See [the simdjson error handling documen
|
||||
You must still validate and consume the values (e.g., call `is_null()`) after calling `type()`.
|
||||
You may also access [the raw JSON string](#general-direct-access-to-the-raw-json-string).
|
||||
For example, the following is a quick and dirty recursive function that verbosely prints the JSON document as JSON. This example also illustrates lifecycle requirements: the `document` instance holds the iterator. The document must remain in scope while you are accessing instances of `value`, `object` and `array`.
|
||||
```c++
|
||||
```cpp
|
||||
void recursive_print_json(ondemand::value element) {
|
||||
bool add_comma;
|
||||
switch (element.type()) {
|
||||
@@ -724,7 +742,7 @@ Let us review these concepts with some additional examples. For simplicity, we o
|
||||
|
||||
The first example illustrates how we can chain operations. In this instance, we repeatedly select keys using the bracket operator (`doc["str"]`) and then finally request a number (using `get_double()`). It is safe to write code in this manner: if any step causes an error, the error status propagates and an exception is thrown at the end. You do not need to constantly check for errors.
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
auto abstract_json = R"(
|
||||
{ "str" : { "123" : {"abc" : 3.14 } } }
|
||||
)"_padded;
|
||||
@@ -738,7 +756,7 @@ an array of objects. We iterate through the objects using a for-loop. Within eac
|
||||
the bracket operator (e.g., `car["make"]`) to select values. We also show how we can iterate through an
|
||||
array, corresponding to the key `tire_pressure`, that is contained inside each object.
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
ondemand::parser parser;
|
||||
auto cars_json = R"( [
|
||||
{ "make": "Toyota", "model": "Camry", "year": 2018, "tire_pressure": [ 40.1, 39.9, 37.7, 40.4 ] },
|
||||
@@ -768,7 +786,7 @@ for (ondemand::object car : parser.iterate(cars_json)) {
|
||||
The previous example had an array of objects, but we can use essentially the same
|
||||
approach with an object of objects.
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
ondemand::parser parser;
|
||||
auto cars_json = R"( {
|
||||
"identifier1":{ "make": "Toyota", "model": "Camry", "year": 2018, "tire_pressure": [ 40.1, 39.9, 37.7, 40.4 ] },
|
||||
@@ -803,7 +821,7 @@ for (ondemand::field key_car : doc.get_object()) {
|
||||
|
||||
The following example illustrates how you may also iterate through object values, effectively visiting all key-value pairs in the object.
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
#include <iostream>
|
||||
#include "simdjson.h"
|
||||
using namespace std;
|
||||
@@ -852,7 +870,7 @@ The C++26 approach is even simpler.
|
||||
|
||||
Suppose you have your own types, such as a `Car` struct:
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
struct Car {
|
||||
std::string make;
|
||||
std::string model;
|
||||
@@ -864,7 +882,7 @@ struct Car {
|
||||
You might want to write code that automatically parses the JSON content to your custom
|
||||
type:
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
padded_string json = R"( [ { "make": "Toyota", "model": "Camry", "year": 2018,
|
||||
"tire_pressure": [ 40.1, 39.9 ] },
|
||||
{ "make": "Kia", "model": "Soul", "year": 2012,
|
||||
@@ -889,7 +907,7 @@ is automatically provided by simdjson if C++20 (and concepts) are available.
|
||||
See [Use `tag_invoke` for custom types](#2-use-tag_invoke-for-custom-types-c20) if you have
|
||||
C++20 support.
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
#if !SIMDJSON_SUPPORTS_CONCEPTS
|
||||
// The code is unnecessary with C++20:
|
||||
template <>
|
||||
@@ -912,7 +930,7 @@ simdjson::ondemand::value::get() noexcept {
|
||||
|
||||
We may then provide support for our `Car` struct:
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
template <>
|
||||
simdjson_inline simdjson_result<Car> simdjson::ondemand::value::get() noexcept {
|
||||
ondemand::object obj;
|
||||
@@ -929,7 +947,7 @@ simdjson_inline simdjson_result<Car> simdjson::ondemand::value::get() noexcept {
|
||||
|
||||
And that is all that is needed! The following code is a complete example:
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
#include "simdjson.h"
|
||||
#include <iostream>
|
||||
#include <vector>
|
||||
@@ -1000,7 +1018,7 @@ Observe that we require an explicit cast (`Car c(val)` instead of `for (Car c :
|
||||
|
||||
If you prefer to avoid exceptions, you may modify the `main` function as follows:
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
int main(void) {
|
||||
padded_string json = R"( [ { "make": "Toyota", "model": "Camry", "year": 2018,
|
||||
"tire_pressure": [ 40.1, 39.9 ] },
|
||||
@@ -1030,7 +1048,7 @@ the `ondemand::document` type. In this instance, we must replace the function wi
|
||||
`simdjson_result<Car> simdjson::ondemand::document::get() &`. The following is a complete
|
||||
example:
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
#include "simdjson.h"
|
||||
#include <iostream>
|
||||
#include <vector>
|
||||
@@ -1110,7 +1128,7 @@ The simdjson library takes advantage of C++20. An immediate benefit
|
||||
is that you can deserialize JSON data directly in standard containers
|
||||
and other standard value types:
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
simdjson::padded_string json = R"({"data" : [1,2,3,4]})"_padded;
|
||||
|
||||
simdjson::ondemand::parser parser;
|
||||
@@ -1120,7 +1138,7 @@ std::vector<uint8_t> array = d["data"].get<std::vector<uint8_t>>();
|
||||
|
||||
Appending to an existing container is just as easy:
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
std::vector<uint32_t> array = {0, 0};
|
||||
|
||||
simdjson::padded_string json = R"({"data" : [1,2,3,4]})"_padded;
|
||||
@@ -1147,7 +1165,7 @@ to 1, otherwise it is set to 0.
|
||||
|
||||
Consider a custom class `Car`:
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
struct Car {
|
||||
std::string make;
|
||||
std::string model;
|
||||
@@ -1164,7 +1182,7 @@ You may support deserializing directly from a JSON value or document to your own
|
||||
by defining a single `tag_invoke` function:
|
||||
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
namespace simdjson {
|
||||
// This tag_invoke MUST be inside simdjson namespace
|
||||
template <typename simdjson_value>
|
||||
@@ -1279,7 +1297,7 @@ By default, we support a wide range of standard templates such as
|
||||
etc. They are handled automatically.
|
||||
|
||||
E.g., you can recover an `std::unique_ptr<Car>` like so:
|
||||
```C++
|
||||
```cpp
|
||||
int main() {
|
||||
auto const json = R"( { "make": "Toyota", "model": "Camry", "year": 2018,
|
||||
"tire_pressure": [ 40.1, 39.9 ] })"_padded;
|
||||
@@ -1293,7 +1311,7 @@ int main() {
|
||||
|
||||
You may also conditionally fill in `std::optional` values.
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
padded_string json =
|
||||
R"( { "car1": { "make": "Toyota", "model": "Camry", "year": 2018,
|
||||
"tire_pressure": [ 40.1, 39.9 ] }
|
||||
@@ -1312,7 +1330,7 @@ You can also deserialize to map-like types with keys that can be constructed
|
||||
from `std::string_view` instances:
|
||||
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
padded_string json =
|
||||
R"( { "car1": { "make": "Toyota", "model": "Camry", "year": 2018,
|
||||
"tire_pressure": [ 40.1, 39.9 ] }
|
||||
@@ -1332,7 +1350,7 @@ Suppose for example that you want to construct an instance of `std::list<Car>`,
|
||||
you also want to filter out any car made by Toyota. You may provide your own
|
||||
`tag_invoke` function:
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
namespace simdjson {
|
||||
// suppose we want to filter out all Toyotas
|
||||
template <typename simdjson_value>
|
||||
@@ -1377,7 +1395,7 @@ Then you can deserialize a type such as `Car` automatically:
|
||||
|
||||
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
struct Car {
|
||||
std::string make;
|
||||
std::string model;
|
||||
@@ -1543,7 +1561,7 @@ Minifying JSON strings without parsing
|
||||
|
||||
In some cases, you may have valid JSON strings that you do not wish to parse but that you wish to minify. That is, you wish to remove all unnecessary spaces. We have a fast function for this purpose (`simdjson::minify(const char * input, size_t length, const char * output, size_t& new_length)`). This function does not validate your content, and it does not parse it. It is much faster than parsing the string and re-serializing it in minified form (`simdjson::minify(parser.parse())`). Usage is relatively simple. You must pass an input pointer with a length parameter, as well as an output pointer and an output length parameter (by reference). The output length parameter is not read, but written to. The output pointer should point to a valid memory region that is as large as the original string length. The input pointer and input length are read, but not written to.
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
// Starts with a valid JSON document as a string.
|
||||
// It does not have to be null-terminated.
|
||||
const char * some_string = "[ 1, 2, 3, 4] ";
|
||||
@@ -1564,7 +1582,7 @@ UTF-8 validation (alone)
|
||||
|
||||
The simdjson library has fast functions to validate UTF-8 strings. They are many times faster than most functions commonly found in libraries. You can use our fast functions, even if you do not care about JSON.
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
const char * some_string = "[ 1, 2, 3, 4] ";
|
||||
size_t length = std::strlen(some_string);
|
||||
bool is_ok = simdjson::validate_utf8(some_string, length);
|
||||
@@ -1581,11 +1599,11 @@ JSON Pointer
|
||||
|
||||
The simdjson library also supports [JSON pointer](https://tools.ietf.org/html/rfc6901) through the `at_pointer()` method, letting you reach further down into the document in a single call. JSON Pointer is supported by both the [DOM approach](https://github.com/simdjson/simdjson/blob/master/doc/dom.md#json-pointer) as well as the On-Demand approach.
|
||||
|
||||
**Note:** The On-Demand implementation of JSON Pointer relies on `find_field` which implies that it does not unescape keys when matching.
|
||||
**Note:** When matching keys, we do a byte-by-byte comparison. We do not unescape keys when matching.
|
||||
|
||||
Consider the following example:
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
auto cars_json = R"( [
|
||||
{ "make": "Toyota", "model": "Camry", "year": 2018, "tire_pressure": [ 40.1, 39.9, 37.7, 40.4 ] },
|
||||
{ "make": "Kia", "model": "Soul", "year": 2012, "tire_pressure": [ 30.1, 31.0, 28.6, 28.7 ] },
|
||||
@@ -1603,7 +1621,7 @@ select the value. If your keys contain the characters '/' or '~', they must be e
|
||||
|
||||
For multiple JSON Pointer queries on a document, one can call `at_pointer` multiple times.
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
auto cars_json = R"( [
|
||||
{ "make": "Toyota", "model": "Camry", "year": 2018, "tire_pressure": [ 40.1, 39.9, 37.7, 40.4 ] },
|
||||
{ "make": "Kia", "model": "Soul", "year": 2012, "tire_pressure": [ 30.1, 31.0, 28.6, 28.7 ] },
|
||||
@@ -1624,7 +1642,7 @@ In most instances, a JSON Pointer is an ASCII string and the keys in a JSON docu
|
||||
are ASCII strings. We support UTF-8 in JSON Pointer, but key values are matched exactly, without unescaping or Unicode normalization. We do a byte-by-byte comparison. The e acute character is
|
||||
considered distinct from its escaped version `\u00E9`. E.g.,
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
const padded_string json = "{\"\\u00E9\":123}"_padded;
|
||||
auto doc = parser.iterate(json);
|
||||
doc.at_pointer("/\\u00E9") == 123; // true
|
||||
@@ -1633,7 +1651,7 @@ doc.at_pointer((const char*)u8"/\u00E9") // returns an error (NO_SUCH_FIELD)
|
||||
|
||||
Note that `at_pointer` calls [`rewind`](#rewind) to reset the parser at the beginning of the document. Hence, it invalidates all previously parsed values, objects and arrays: make sure to consume the values between each call to `at_pointer`. Consider the following example where one wants to store each object from the JSON into a vector of `struct car_type`:
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
struct car_type {
|
||||
std::string make;
|
||||
std::string model;
|
||||
@@ -1676,7 +1694,7 @@ for (int i = 0; i < 3; i++) {
|
||||
|
||||
Furthermore, `at_pointer` calls `rewind` at the beginning of the call (i.e. the document is not reset after `at_pointer`). Consider the following example,
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
auto json = R"( {
|
||||
"k0": 27,
|
||||
"k1": [13,26],
|
||||
@@ -1696,7 +1714,7 @@ be represented as `value` instances. You can check that a document is a scalar w
|
||||
JSONPath
|
||||
------------
|
||||
|
||||
The simdjson library supports a subset of [JSONPath](https://datatracker.ietf.org/doc/html/draft-normington-jsonpath-00) through the `at_path()` method, allowing you to reach further into the document in a single call. The subset of JSONPath that is implemented is the subset that is trivially convertible into the JSON Pointer format, using `.` to access a field and `[]` to access a specific index.
|
||||
The simdjson library supports a subset of [JSONPath](https://www.rfc-editor.org/rfc/rfc9535) (RFC 9535) through the `at_path()` method, allowing you to reach further into the document in a single call. The subset of JSONPath that is implemented is the subset that is trivially convertible into the JSON Pointer format, using `.` to access a field and `[]` to access a specific index.
|
||||
|
||||
This implementation relies on `at_path()` converting its argument to JSON Pointer and then calling `at_pointer`, which makes use of
|
||||
[`rewind`](#rewind) to reset the parser at the beginning of the document. Hence, it invalidates all previously parsed values, objects
|
||||
@@ -1704,7 +1722,7 @@ This implementation relies on `at_path()` converting its argument to JSON Pointe
|
||||
|
||||
Consider the following example:
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
auto cars_json = R"( [
|
||||
{ "make": "Toyota", "model": "Camry", "year": 2018, "tire_pressure": [ 40.1, 39.9, 37.7, 40.4 ] },
|
||||
{ "make": "Kia", "model": "Soul", "year": 2012, "tire_pressure": [ 30.1, 31.0, 28.6, 28.7 ] },
|
||||
@@ -1717,7 +1735,7 @@ cout << cars.at_path("[0].tire_pressure[1]") << endl; // Prints 39.9
|
||||
|
||||
A call to `at_path(json_path)` can result in any of the errors that are returned by the `at_pointer` method and if the conversion of `json_path` to JSON Pointer fails, it will lead to an `simdjson::INVALID_JSON_POINTER`error.
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
auto cars_json = R"( [
|
||||
{ "make": "Toyota", "model": "Camry", "year": 2018, "tire_pressure": [ 40.1, 39.9, 37.7, 40.4 ] },
|
||||
{ "make": "Kia", "model": "Soul", "year": 2012, "tire_pressure": [ 30.1, 31.0, 28.6, 28.7 ] },
|
||||
@@ -1734,7 +1752,7 @@ are ASCII strings. We support UTF-8 within a JSONPath expression, but key values
|
||||
matched exactly, without unescaping or Unicode normalization. We do a byte-by-byte comparison.
|
||||
The e acute character is considered distinct from its escaped version `\u00E9`. E.g.,
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
const padded_string json = "{\"\\u00E9\":123}"_padded;
|
||||
auto doc = parser.iterate(json);
|
||||
doc.at_path(".\\u00E9") == 123; // true
|
||||
@@ -1744,7 +1762,7 @@ doc.at_path((const char*)u8".\u00E9") // returns an error (NO_SUCH_FIELD)
|
||||
|
||||
We also support the `$` prefix. When you start a JSONPath expression with $, you are indicating that the path starts from the root of the JSON document. E.g.,
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
auto json = R"( { "c" :{ "foo": { "a": [ 10, 20, 30 ] }}, "d": { "foo2": { "a": [ 10, 20, 30 ] }} , "e": 120 })"_padded;
|
||||
ondemand::parser parser;
|
||||
ondemand::document doc = parser.iterate(json);
|
||||
@@ -1753,6 +1771,89 @@ int64_t x = obj.at_path("$.c.foo.a[1]"); // 20
|
||||
x = obj.at_path("$.d.foo2.a.2"); // 30
|
||||
```
|
||||
|
||||
## Using `at_path_with_wildcard` for JSONPath Queries (On-Demand)
|
||||
|
||||
The `at_path_with_wildcard` function in simdjson extends the JSONPath querying capabilities by supporting wildcard expressions (`*`) in JSON paths. This allows users to retrieve multiple elements from a JSON document in a single query. For example, you can use `$.address.*` to fetch all fields within the `address` object or `$.phoneNumbers[*].numbers[*]` to retrieve all phone numbers across multiple objects in an array.
|
||||
|
||||
The `*` wildcard matches all elements at a specific level. For instance, `$.address.*` retrieves all key-value pairs in the `address` object, while `$.*.streetAddress` fetches all `streetAddress` fields across objects at the root level. You can combine wildcards with array indexing. For example, `$.phoneNumbers[*].numbers[1]` retrieves the second number from each `numbers` array in the `phoneNumbers` array. If no elements match the wildcard query, the function returns an empty result. For instance, querying `$.empty_object.*` or `$.empty_array.*` will yield an empty set.
|
||||
|
||||
### Example Usage
|
||||
|
||||
Here is an example demonstrating the use of `at_path_with_wildcard`:
|
||||
|
||||
```cpp
|
||||
simdjson::padded_string json_string = R"(
|
||||
{
|
||||
"firstName": "John",
|
||||
"lastName": "doe",
|
||||
"age": 26,
|
||||
"address": {
|
||||
"streetAddress": "naist street",
|
||||
"city": "Nara",
|
||||
"postalCode": "630-0192"
|
||||
},
|
||||
"phoneNumbers": [
|
||||
{
|
||||
"type": "iPhone",
|
||||
"numbers": ["0123-4567-8888", "0123-4567-8788"]
|
||||
},
|
||||
{
|
||||
"type": "home",
|
||||
"numbers": ["0123-4567-8910"]
|
||||
}
|
||||
]
|
||||
})"_padded;
|
||||
|
||||
ondemand::parser parser;
|
||||
ondemand::document doc = parser.iterate(json_string);
|
||||
|
||||
// Fetch all fields in the address object
|
||||
std::vector<ondemand::value> values;
|
||||
auto error = doc.at_path_with_wildcard("$.address.*").get(values);
|
||||
if (!error) {
|
||||
for (auto value : values) {
|
||||
std::string_view field;
|
||||
if (value.get(field) == SUCCESS) {
|
||||
std::cout << field << std::endl;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Fetch all phone numbers
|
||||
error = doc.at_path_with_wildcard("$.phoneNumbers[*].numbers[*]").get(values);
|
||||
if (!error) {
|
||||
for (auto value : values) {
|
||||
std::string_view number;
|
||||
if (value.get(number) == SUCCESS) {
|
||||
std::cout << number << std::endl;
|
||||
}
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
This function is particularly useful for extracting data from complex JSON structures with nested arrays and objects. By leveraging wildcards, you can simplify your queries and reduce the need for multiple iterations.
|
||||
|
||||
## Compile-Time JSONPath and JSON Pointer (C++26 Reflection)
|
||||
|
||||
The simdjson library provides **compile-time validated** JSONPath and JSON Pointer accessors when using C++26 Static Reflection. These accessors validate paths against struct definitions at compile time and generate optimized code with zero runtime overhead. In some cases, we find that it is much faster. Furthermore, it is safer in the sense that the expression
|
||||
is validated at compile-time.
|
||||
|
||||
**Requirements:** C++26 compiler with P2996 reflection support and `-DSIMDJSON_STATIC_REFLECTION=ON` build flag.
|
||||
|
||||
```cpp
|
||||
ondemand::parser parser;
|
||||
auto doc = parser.iterate(json);
|
||||
|
||||
// Without validation - path parsed at compile time only
|
||||
std::string_view city;
|
||||
result = ondemand::json_path::at_path_compiled<".address.city">(doc);
|
||||
result.get(city);
|
||||
```
|
||||
|
||||
We further provide type-validation so that you can check that the types are as you expect.
|
||||
|
||||
**See [Compile-Time Accessors](compile_time_accessors.md) for complete documentation.**
|
||||
|
||||
Error handling
|
||||
--------------
|
||||
|
||||
@@ -1761,7 +1862,7 @@ Error handling with exception and a single try/catch clause makes the code simpl
|
||||
The entire simdjson API is usable with and without exceptions. All simdjson APIs that can fail return `simdjson_result<T>`, which is a <value, error_code>
|
||||
pair. You can retrieve the value with .get() without generating an exception, like so:
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
ondemand::document doc;
|
||||
auto error = parser.iterate(json).get(doc);
|
||||
if(error) { std::cerr << simdjson::error_message(error); exit(1); }
|
||||
@@ -1798,7 +1899,7 @@ set of warnings: they can identify variables that are written to but never other
|
||||
Let us illustrate with an example where we try to access a number that is not valid (`3.14.1`).
|
||||
If we want to proceed without throwing and catching exceptions, we can do so as follows:
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
bool simple_error_example() {
|
||||
ondemand::parser parser;
|
||||
auto json = R"({"bad number":3.14.1 })"_padded;
|
||||
@@ -1820,7 +1921,7 @@ Observe how we verify the error variable before accessing the retrieved number (
|
||||
|
||||
The equivalent with exception handling might look as follows.
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
bool simple_error_example_except() {
|
||||
TEST_START();
|
||||
ondemand::parser parser;
|
||||
@@ -1861,7 +1962,7 @@ We can write a "quick start" example where we attempt to parse the following JSO
|
||||
Our program loads the file, selects value corresponding to key `"search_metadata"` which expected to be an object, and then
|
||||
it selects the key `"count"` within that object.
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
#include <iostream>
|
||||
#include "simdjson.h"
|
||||
|
||||
@@ -1893,7 +1994,7 @@ triggering exceptions. To do this, we use `["statuses"].at(0)["id"]`. We break t
|
||||
Observe how we use the `at` method when querying an index into an array, and not the bracket operator.
|
||||
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
#include <iostream>
|
||||
#include "simdjson.h"
|
||||
|
||||
@@ -1919,7 +2020,7 @@ to iterate through the values of an array. We deliberately forbid this usage to
|
||||
|
||||
This is how the example in "Using the parsed JSON" could be written using only error code checking (without exceptions):
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
bool parse() {
|
||||
ondemand::parser parser;
|
||||
auto cars_json = R"( [
|
||||
@@ -1976,7 +2077,7 @@ bool parse() {
|
||||
For safety, you should only use our ondemand instances (e.g., `ondemand::object`)
|
||||
after you have initialized them and checked that there is no error:
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
ondemand::object car; // invalid until the get() succeeds
|
||||
// the `car` instance should not use used before it is initialized
|
||||
error = car_value.get_object().get(car);
|
||||
@@ -1989,7 +2090,7 @@ after you have initialized them and checked that there is no error:
|
||||
|
||||
The following examples illustrates how to iterate through the content of an object without
|
||||
having to handle exceptions.
|
||||
```c++
|
||||
```cpp
|
||||
auto json = R"({"k\u0065y": 1})"_padded;
|
||||
ondemand::parser parser;
|
||||
ondemand::document doc;
|
||||
@@ -2025,7 +2126,7 @@ target_compile_definitions(simdjson PUBLIC SIMDJSON_EXCEPTIONS=OFF)
|
||||
|
||||
Users more comfortable with an exception flow may choose to directly cast the `simdjson_result<T>` to the desired type:
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
simdjson::ondemand::document doc = parser.iterate(json); // Throws an exception if there was an error!
|
||||
```
|
||||
|
||||
@@ -2035,7 +2136,7 @@ program from continuing if there was an error.
|
||||
|
||||
If one is willing to trigger exceptions, it is possible to write simpler code:
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
#include <iostream>
|
||||
#include "simdjson.h"
|
||||
|
||||
@@ -2052,7 +2153,7 @@ int main(void) {
|
||||
|
||||
You can do handle errors gracefully as well...
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
#include <iostream>
|
||||
#include "simdjson.h"
|
||||
int main(void) {
|
||||
@@ -2079,7 +2180,7 @@ When the input was a `padding_string` or another null-terminated source, then yo
|
||||
use the `const char *` pointer as a C string. As an example, consider the following
|
||||
example where we used the exception-free simdjson interface:
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
auto broken_json = R"( {"double": 13.06, false, "integer": -343} )"_padded; // Missing key
|
||||
ondemand::parser parser;
|
||||
auto doc = parser.iterate(broken_json);
|
||||
@@ -2099,7 +2200,7 @@ if (error) {
|
||||
|
||||
You may also use `current_location()` with exceptions as follows:
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
auto broken_json = R"( {"double": 13.06, false, "integer": -343} )"_padded;
|
||||
ondemand::parser parser;
|
||||
ondemand::document doc = parser.iterate(broken_json);
|
||||
@@ -2116,7 +2217,7 @@ had to go through a value without a key before (`false`), a `TAPE_ERROR` error i
|
||||
The pointer returned by the `current_location()` method then points at the location of the error. The `current_location()` may also be used when the error is triggered
|
||||
by a user action, even if the JSON input is valid. Consider the following example:
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
auto json = R"( [1,2,3] )"_padded;
|
||||
ondemand::parser parser;
|
||||
auto doc = parser.iterate(json);
|
||||
@@ -2131,7 +2232,7 @@ if (error) {
|
||||
If the location is invalid (i.e. at the end of a document), the `current_location()`
|
||||
methods returns an `OUT_OF_BOUNDS` error. For example:
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
auto json = R"( [1,2,3] )"_padded;
|
||||
ondemand::parser parser;
|
||||
auto doc = parser.iterate(json);
|
||||
@@ -2147,7 +2248,7 @@ then the document has more content.
|
||||
Finally, the `current_location()` method may also be used even when no exceptions/errors
|
||||
are thrown. This can be helpful for users that want to know the current state of iteration during parsing. For example:
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
auto json = R"( [[1,2,3], -23.4, {"key": "value"}, true] )"_padded;
|
||||
ondemand::parser parser;
|
||||
auto doc = parser.iterate(json);
|
||||
@@ -2180,7 +2281,7 @@ content.
|
||||
|
||||
Example 1.
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
auto json = R"([1, 2] foo ])"_padded;
|
||||
ondemand::parser parser;
|
||||
ondemand::document doc = parser.iterate(json);
|
||||
@@ -2223,7 +2324,7 @@ that you have created so far (including unescaped strings).
|
||||
In the following example, we print on the screen the number of cars in the JSON input file
|
||||
before printout the data.
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
ondemand::parser parser;
|
||||
auto cars_json = R"( [
|
||||
{ "make": "Toyota", "model": "Camry", "year": 2018, "tire_pressure": [ 40.1, 39.9, 37.7, 40.4 ] },
|
||||
@@ -2271,7 +2372,7 @@ individual document must be no larger than 4 GB.
|
||||
|
||||
Here is an example:
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
auto json = R"({ "foo": 1 } { "foo": 2 } { "foo": 3 } )"_padded;
|
||||
ondemand::parser parser;
|
||||
ondemand::document_stream docs = parser.iterate_many(json);
|
||||
@@ -2293,7 +2394,7 @@ The `iterate_many` function can also take an optional parameter `size_t batch_si
|
||||
|
||||
The following toy examples illustrates how to get capacity errors. It is an artificial example since you should never use a `batch_size` of 50 bytes (it is far too small).
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
// We are going to set the capacity to 50 bytes which means that we cannot
|
||||
// loading a document longer than 50 bytes. The first few documents are small,
|
||||
// but the last one is large. We will get an error at the last document.
|
||||
@@ -2353,7 +2454,7 @@ methods appropriately. In particular, a valid JSON number has no leading and no
|
||||
numbers (although you have access to the raw string with the `raw_json_token()` method, see [General direct access to the raw JSON string](#general-direct-access-to-the-raw-json-string)
|
||||
). As an example, suppose we have the following JSON text:
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
auto json =
|
||||
{
|
||||
"ticker":{
|
||||
@@ -2387,7 +2488,7 @@ auto json =
|
||||
|
||||
Now, suppose that a user wants to get the time stamp from the `timestampstr` key. One could do the following:
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
ondemand::parser parser;
|
||||
auto doc = parser.iterate(json);
|
||||
uint64_t time = doc.at_pointer("/timestampstr").get_uint64_in_string();
|
||||
@@ -2396,7 +2497,7 @@ std::cout << time << std::endl; // Prints 1399490941
|
||||
|
||||
Another thing a user might want to do is extract the `markets` array and get the market name, price and volume. Here is one way to do so:
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
ondemand::parser parser;
|
||||
auto doc = parser.iterate(json);
|
||||
|
||||
@@ -2418,7 +2519,7 @@ Market: btce Price: 432.89 Volume: 8561.06
|
||||
|
||||
Finally, here is an example dealing with errors where the user wants to convert the string `"Infinity"`(`"change"` key) to a float with infinity value.
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
ondemand::parser parser;
|
||||
auto doc = parser.iterate(json);
|
||||
// Get "change"/"Infinity" key/value pair
|
||||
@@ -2487,7 +2588,7 @@ The `get_number()` function is designed with performance in mind. When calling `
|
||||
|
||||
|
||||
Consider the following example:
|
||||
```C++
|
||||
```cpp
|
||||
ondemand::parser parser;
|
||||
padded_string docdata = R"([1.0, 3, 1, 3.1415,-13231232,9999999999999999999])"_padded;
|
||||
ondemand::document doc = parser.iterate(docdata);
|
||||
@@ -2534,7 +2635,7 @@ unsigned integers. Calling `get_number_type()` on the values returns `ondemand::
|
||||
You can try to represent these big integers as 64-bit floating-point numbers, though you typically lose
|
||||
precision in the process (as illustrated in the example).
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
ondemand::parser parser;
|
||||
padded_string docdata = R"([-9223372036854775809, 18446744073709551617, 99999999999999999999999 ])"_padded;
|
||||
double dexpected[] = {-9223372036854775808.0, 18446744073709551616.0, 1e23};
|
||||
@@ -2557,7 +2658,7 @@ This program might print:
|
||||
You may get access to the underlying string representing the big integer with
|
||||
`raw_json_token()` and you may parse the resulting number strings using your own parser.
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
ondemand::parser parser;
|
||||
padded_string docdata = R"([-9223372036854775809, 18446744073709551617, 99999999999999999999999 ])"_padded;
|
||||
ondemand::document doc = parser.iterate(docdata);
|
||||
@@ -2596,7 +2697,7 @@ you should ensure that you have sufficient memory space: the total size of the s
|
||||
`simdjson::SIMDJSON_PADDING` bytes. The following example illustrates how we can unescape
|
||||
JSON string to a user-provided buffer:
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
auto json = R"( {"name": "Jack The Ripper \u0033"} )"_padded;
|
||||
// We create a buffer large enough to store all strings we need:
|
||||
std::unique_ptr<uint8_t[]> buffer(new uint8_t[json.size() + simdjson::SIMDJSON_PADDING]);
|
||||
@@ -2620,7 +2721,7 @@ purpose. It provides a view on the key, including the starting quote character,
|
||||
and everything up to the next `:` character after the final quote character. E.g.,
|
||||
if the key is `"name"` then `key_raw_json_token()` returns a `std::string_view` which
|
||||
begins with `"name"` and may containing trailing white-space characters.
|
||||
```C++
|
||||
```cpp
|
||||
auto json = R"( {"name" : "Jack The Ripper \u0033"} )"_padded;
|
||||
ondemand::parser parser;
|
||||
ondemand::document doc = parser.iterate(json);
|
||||
@@ -2643,7 +2744,7 @@ The library makes this possible by providing a `raw_json_token` method which ret
|
||||
a `std::string_view` instance containing the value as a string which you may then
|
||||
parse as you see fit.
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
simdjson::ondemand::parser parser;
|
||||
simdjson::padded_string docdata = R"({"value":12321323213213213213213213213211223})"_padded;
|
||||
simdjson::ondemand::document doc = parser.iterate(docdata);
|
||||
@@ -2656,7 +2757,7 @@ The `raw_json_token` method even works when the JSON value is a string. In such
|
||||
will return the complete string with the quotes and with eventual escaped sequences as in the
|
||||
source document.
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
simdjson::ondemand::parser parser;
|
||||
simdjson::padded_string docdata = R"({"value":"12321323213213213213213213213211223"})"_padded;
|
||||
simdjson::ondemand::document doc = parser.iterate(docdata);
|
||||
@@ -2704,7 +2805,7 @@ If your value is an array or an object, `raw_json_token()` returns effectively a
|
||||
character (`[`) or (`}`) which is not very useful. For arrays and objects, we have another
|
||||
method called `raw_json()` which consumes (traverses) the array or the object.
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
simdjson::ondemand::parser parser;
|
||||
simdjson::padded_string docdata = R"({"value":123})"_padded;
|
||||
simdjson::ondemand::document doc = parser.iterate(docdata);
|
||||
@@ -2713,7 +2814,7 @@ string_view token = obj.raw_json(); // gives you `{"value":123}`
|
||||
```
|
||||
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
simdjson::ondemand::parser parser;
|
||||
simdjson::padded_string docdata = R"([1,2,3])"_padded;
|
||||
simdjson::ondemand::document doc = parser.iterate(docdata);
|
||||
@@ -2724,7 +2825,7 @@ string_view token = arr.raw_json(); // gives you `[1,2,3]`
|
||||
Because `raw_json()` consumes to object or the array, if you want to both have
|
||||
access to the raw string, and also use the array or object, you should call `reset()`.
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
simdjson::ondemand::parser parser;
|
||||
simdjson::padded_string docdata = R"({"value":123})"_padded;
|
||||
simdjson::ondemand::document doc = parser.iterate(docdata);
|
||||
@@ -2740,7 +2841,7 @@ value is an array or an object. Otherwise, it acts as `raw_json_token()`.
|
||||
It is useful if you do not care for the type of the value and just wants a
|
||||
string representation.
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
auto json = R"( [1,2,"fds", {"a":1}, [1,344]] )"_padded;
|
||||
ondemand::parser parser;
|
||||
ondemand::document doc = parser.iterate(json);
|
||||
@@ -2751,7 +2852,7 @@ string representation.
|
||||
}
|
||||
```
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
auto json = R"( {"key1":1,"key2":2,"key3":"fds", "key4":{"a":1}, "key5":[1,344]} )"_padded;
|
||||
ondemand::parser parser;
|
||||
ondemand::document doc = parser.iterate(json);
|
||||
@@ -2799,7 +2900,7 @@ However, they are cases where you need to store a string result in a `std::strin
|
||||
instance. You can do so with a templated version of the `to_string()` method which takes as
|
||||
a parameter a reference to a `std::string`.
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
auto json = R"({
|
||||
"name": "Daniel",
|
||||
"age": 42
|
||||
@@ -2813,7 +2914,7 @@ a parameter a reference to a `std::string`.
|
||||
|
||||
The same routine can be written without exceptions handling:
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
std::string name;
|
||||
auto error = doc["name"].get_string(name);
|
||||
if (error) { /* handle error */ }
|
||||
@@ -2826,7 +2927,7 @@ only consume a JSON string once.
|
||||
Because `get_string()` is a template that requires a type that can be assigned a `std::string`, you
|
||||
can use it with features such as `std::optional`:
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
auto json = R"({ "foo1": "3.1416" } )"_padded;
|
||||
ondemand::parser parser;
|
||||
ondemand::document doc = parser.iterate(json);
|
||||
@@ -2907,7 +3008,7 @@ For simplicity, we do not include full error support: this code would throw exce
|
||||
|
||||
* Example 1: ZuluBBox
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
struct ZuluBBox {
|
||||
double xmin;
|
||||
double ymin;
|
||||
@@ -3011,7 +3112,7 @@ bool example() {
|
||||
|
||||
* Example 2: Demos
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
bool example() {
|
||||
auto json = R"+( {
|
||||
"5f08a730b280e54fd1e75a7046b93fdc": {
|
||||
@@ -3087,7 +3188,7 @@ bool example() {
|
||||
|
||||
* Example 3: CRT
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
|
||||
bool example() {
|
||||
padded_string padded_input_json = R"([
|
||||
@@ -3164,7 +3265,7 @@ bool example() {
|
||||
|
||||
* Example 4: Passing an array to a function
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
|
||||
#include "simdjson.h"
|
||||
#include <iostream>
|
||||
@@ -3259,13 +3360,13 @@ Performance tips
|
||||
}
|
||||
```
|
||||
- If possible, refer to each object and array in your code once. For example, the following code repeatedly refers to the `"data"` key to create an object...
|
||||
```C++
|
||||
```cpp
|
||||
std::string_view make = o["data"]["make"];
|
||||
std::string_view model = o["data"]["model"];
|
||||
std::string_view year = o["data"]["year"];
|
||||
```
|
||||
We expect that it is more efficient to access the `"data"` key once:
|
||||
```C++
|
||||
```cpp
|
||||
simdjson::ondemand::object data = o["data"];
|
||||
std::string_view model = data["model"];
|
||||
std::string_view year = data["year"];
|
||||
|
||||
@@ -1,5 +1,8 @@
|
||||
We take our documentation seriously. Please start reading the documentation before you attempt to use simdjson. We hope you will enjoy reading us.
|
||||
|
||||
* Basics: https://github.com/simdjson/simdjson/blob/master/doc/basics.md is an overview of how to use simdjson and its APIs.
|
||||
* iterate_many: https://github.com/simdjson/simdjson/blob/master/doc/iterate_many.md describes an interface providing features to work with files or streams containing multiple small JSON documents. As fast and convenient as possible.
|
||||
* Performance: https://github.com/simdjson/simdjson/blob/master/doc/performance.md shows some more advanced scenarios and how to tune for them.
|
||||
* [Basics](doc/basics.md) is an overview of how to use simdjson and its APIs.
|
||||
* [Builder](doc/builder.md) is an overview of how to efficiently write JSON strings using simdjson.
|
||||
* [Performance](doc/performance.md) shows some more advanced scenarios and how to tune for them.
|
||||
* [Implementation Selection](doc/implementation-selection.md) describes runtime CPU detection and
|
||||
how you can work with it.
|
||||
* [API](https://simdjson.github.io/simdjson/) contains the automatically generated API documentation.
|
||||
|
||||
@@ -14,6 +14,7 @@ speed and high convenience.
|
||||
* [C++26 static reflection](#c--26-static-reflection)
|
||||
+ [Without `string_buffer` instance](#without--string-buffer--instance)
|
||||
+ [Without `string_buffer` instance but with explicit error handling](#without--string-buffer--instance-but-with-explicit-error-handling)
|
||||
+ [Pretty formatted (fractured JSON)](#pretty-formatted-fractured-json)
|
||||
|
||||
Overview: string_builder
|
||||
---------------------------
|
||||
@@ -60,7 +61,7 @@ The later method (`view()`) is recommended. For performance reasons, we expect
|
||||
Example: string_builder
|
||||
---------------------------
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
struct Car {
|
||||
std::string make;
|
||||
std::string model;
|
||||
@@ -332,7 +333,7 @@ pattern:
|
||||
|
||||
### Customization
|
||||
|
||||
If you want to serialize a value in a custome way, you can do it with a
|
||||
If you want to serialize a value in a custom way, you can do it with a
|
||||
`tag_invoke` specialization like the following example which will map
|
||||
the year attribute to a string.
|
||||
|
||||
@@ -363,4 +364,50 @@ void tag_invoke(serialize_tag, builder_type &builder, const Car& car) {
|
||||
}
|
||||
|
||||
} // namespace simdjson
|
||||
```
|
||||
```
|
||||
|
||||
### Pretty formatted (fractured JSON)
|
||||
|
||||
In some instances, you may want your JSON to be more readable. For this pupose, we also
|
||||
support the Fractured JSON standard.
|
||||
|
||||
```Cpp
|
||||
TableTestData data{
|
||||
{{1, "Alice", true}, {2, "Bob", false}, {3, "Carol", true}, {4, "Dave", false}}
|
||||
};
|
||||
|
||||
fractured_json_options opts;
|
||||
opts.enable_table_format = true;
|
||||
opts.min_table_rows = 3;
|
||||
|
||||
std::string formatted = simdjson::to_fractured_json_string(data, opts);
|
||||
```
|
||||
|
||||
The result might be as follows.
|
||||
|
||||
```json
|
||||
{
|
||||
"records": [
|
||||
{ "active": true , "id": 1, "name": "Alice" },
|
||||
{ "active": false, "id": 2, "name": "Bob" },
|
||||
{ "active": true , "id": 3, "name": "Carol" },
|
||||
{ "active": false, "id": 4, "name": "Dave" }
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
The `fractured_json_options` struct allows you to customize the formatting behavior. It includes the following options:
|
||||
|
||||
- `max_total_line_length` (default: 120): Maximum total characters per line. Content exceeding this will be expanded to multiple lines.
|
||||
- `max_inline_length` (default: 80): Maximum length for inlined elements. Simple arrays/objects shorter than this may be rendered inline.
|
||||
- `max_inline_complexity` (default: 2): Maximum nesting depth for inline rendering. Elements with complexity exceeding this will be expanded. Complexity 0 = scalar, 1 = flat array/object, 2 = one level of nesting.
|
||||
- `max_compact_array_complexity` (default: 1): Maximum complexity for compact array formatting. Arrays with elements of this complexity or less may have multiple items per line.
|
||||
- `indent_spaces` (default: 4): Number of spaces per indentation level.
|
||||
- `enable_table_format` (default: true): Enable tabular formatting for arrays of similar objects. When enabled, arrays of objects with identical keys are formatted as aligned tables.
|
||||
- `min_table_rows` (default: 3): Minimum number of rows to trigger table mode.
|
||||
- `table_similarity_threshold` (default: 0.8): Similarity threshold for table detection. Objects must share at least this fraction of keys to be formatted as a table.
|
||||
- `enable_compact_multiline` (default: true): Enable compact multiline arrays. When enabled, arrays of simple elements may have multiple items per line.
|
||||
- `max_items_per_line` (default: 10): Maximum array items per line in compact mode.
|
||||
- `simple_bracket_padding` (default: true): Add space inside brackets for simple containers. When true: `{ "key": "value" }`, when false: `{"key": "value"}`.
|
||||
- `colon_padding` (default: true): Add space after colons. When true: `"key": "value"`, when false: `"key":"value"`.
|
||||
- `comma_padding` (default: true): Add space after commas in inline content. When true: `[1, 2, 3]`, when false: `[1,2,3]`.
|
||||
@@ -0,0 +1,227 @@
|
||||
# Parse json at compile time
|
||||
* [Introduction](#introduction)
|
||||
* [Example](#example)
|
||||
* [Concepts](#concepts)
|
||||
* [Loading from disk](#loading-from-disk)
|
||||
* [Limitations (compile-time errors)](#limitations-compile-time-errors)
|
||||
|
||||
|
||||
## Introduction
|
||||
In some instances, you may want to configure your software at compile-time with a JSON document.
|
||||
Maybe you have a single code base but many different possible configurations, all resulting in
|
||||
different software. For example, you might be programming robots, using the same software, but
|
||||
different robot configurations.
|
||||
|
||||
To achieve the desired result, you have a few options. You may start the software and parser a
|
||||
JSON file at runtime. Or you might convert your JSON data into C++ code that you can compile with
|
||||
your software.
|
||||
|
||||
With C++26, there is another way: parse the JSON file along with your C++ code. In this manner,
|
||||
the JSON data becomes native C++ data.
|
||||
|
||||
The simdjson library supports parsing JSON documents at compile time if you have C++26 support. To
|
||||
activate C++26 reflection support, you can compile
|
||||
your code with the `SIMDJSON_STATIC_REFLECTION` macro set:
|
||||
|
||||
```cpp
|
||||
#define SIMDJSON_STATIC_REFLECTION 1
|
||||
//...
|
||||
#include "simdjson.h"
|
||||
```
|
||||
|
||||
The `simdjson::compile_time::parse_json` function parses a JSON document at **compile time** and returns a `constexpr` structure reflecting its content. We support the full range of JSON values, which are mapped to C++ types as in
|
||||
the following table.
|
||||
|
||||
|
||||
| JSON type | C++ type |
|
||||
|----------------|----------------------------------|
|
||||
| object | anonymous struct |
|
||||
| array | `std::array<T, N>` (homogeneous) |
|
||||
| string | `const char*` (UTF-8) |
|
||||
| number | `int64_t`, `uint64_t`, `double` |
|
||||
| `true`/`false` | `bool` |
|
||||
| `null` | `std::nullptr_t` |
|
||||
|
||||
## Example
|
||||
|
||||
Suppose you want to parse the following JSON document:
|
||||
|
||||
```cpp
|
||||
{
|
||||
"port": 8080,
|
||||
"host": "localhost",
|
||||
"debug": true
|
||||
}
|
||||
```
|
||||
|
||||
**Reminder**: In C++, `R"( )"` allows us to write multi-line strings with unescaped quotes.
|
||||
|
||||
|
||||
You can do so, at compile-time, as follows:
|
||||
|
||||
|
||||
```cpp
|
||||
constexpr auto cfg = R"(
|
||||
|
||||
{
|
||||
"port": 8080,
|
||||
"host": "localhost",
|
||||
"debug": true
|
||||
}
|
||||
|
||||
)"_json;
|
||||
|
||||
// cfg.port == 8080
|
||||
// std::string_view(cfg.host) == "localhost"
|
||||
// cfg.debug == true
|
||||
```
|
||||
|
||||
You can nest objects and arrays:
|
||||
|
||||
```cpp
|
||||
constexpr auto data = R"(
|
||||
|
||||
{
|
||||
"servers": [
|
||||
{"host": "s1", "port": 3000},
|
||||
{"host": "s2", "port": 3001}
|
||||
]
|
||||
}
|
||||
|
||||
)"_json;
|
||||
|
||||
|
||||
// data.servers.size() == 2
|
||||
// std::string_view(data.servers[0].host) == "s1"
|
||||
```
|
||||
|
||||
Top-level arrays are allowed:
|
||||
|
||||
```cpp
|
||||
constexpr auto arr = R"(
|
||||
|
||||
[1, 2, 3]
|
||||
|
||||
)"_json;
|
||||
static_assert(arr.size() == 3);
|
||||
static_assert(arr[1] == 2);
|
||||
```
|
||||
|
||||
|
||||
## Concepts
|
||||
|
||||
Given that the parsed data is made of structures that depend on the JSON input, you might
|
||||
want to check that it conforms to your expectation. You can do so with concepts.
|
||||
|
||||
Let us consider this example:
|
||||
|
||||
```cpp
|
||||
constexpr auto config = R"(
|
||||
|
||||
[
|
||||
{ "name": "Alice", "age": 30 },
|
||||
{ "name": "Bob", "age": 25 },
|
||||
{ "name": "Charlie", "age": 35 }
|
||||
]
|
||||
|
||||
)"_json;
|
||||
```
|
||||
|
||||
You might want to ensure that the result is an array of persons. You can define your
|
||||
expectation with concepts like so:
|
||||
|
||||
```cpp
|
||||
template <typename T>
|
||||
concept person = requires(T p) {
|
||||
std::string_view(p.name); // has name field convertible to string_view
|
||||
p.age; // has age field
|
||||
requires std::is_integral_v<decltype(p.age)>; // age is integral
|
||||
};
|
||||
|
||||
/**
|
||||
* Concept to validate that a type is an array of person objects
|
||||
*/
|
||||
template <typename T>
|
||||
concept array_of_person = requires(T arr) {
|
||||
arr.size(); // has size method
|
||||
arr[0]; // can access elements with []
|
||||
requires person<decltype(arr[0])>; // elements satisfy person concept
|
||||
};
|
||||
```
|
||||
|
||||
And then a simple static assert with `decltype` is sufficient to check that the expectation is met:
|
||||
|
||||
```cpp
|
||||
constexpr auto config = R"(
|
||||
|
||||
[
|
||||
{ "name": "Alice", "age": 30 },
|
||||
{ "name": "Bob", "age": 25 },
|
||||
{ "name": "Charlie", "age": 35 }
|
||||
]
|
||||
|
||||
)"_json;
|
||||
|
||||
|
||||
// Validate that the array satisfies the array_of_person concept
|
||||
static_assert(array_of_person<decltype(config)>);
|
||||
```
|
||||
|
||||
|
||||
|
||||
## Loading from disk
|
||||
|
||||
In practice, you may have a JSON file, say `json_data` that you want to parse
|
||||
at compile time. You may do so as follows.
|
||||
|
||||
```c++
|
||||
constexpr const char json_data[] = {
|
||||
#embed "test.json"
|
||||
, 0
|
||||
};
|
||||
|
||||
constexpr auto json = simdjson::compile_time::parse_json<json_data>();
|
||||
```
|
||||
|
||||
|
||||
## Limitations (compile-time errors)
|
||||
|
||||
We have a few limitations which trigger compile-time errors if violated.
|
||||
|
||||
|
||||
- Only JSON objects and arrays are supported at the top level (no primitives).
|
||||
We will lift this limitation in the future.
|
||||
- Strings are represented using the `const char*` in UTF-8, but they must not
|
||||
contain embedded nulls. We would prefer to represent them as std::string or
|
||||
std::string_view, and hope to do so in the future.
|
||||
- Heterogeneous arrays are not supported yet. E.g., you need to have arrays of
|
||||
all integers, or all strings, all floats, all compatible objects, etc.
|
||||
For example, the following is accepted:
|
||||
```json
|
||||
[
|
||||
{ "name": "Alice", "age": 30 },
|
||||
{ "name": "Bob", "age": 25 },
|
||||
{ "name": "Charlie", "age": 35 }
|
||||
]
|
||||
```
|
||||
but the following is not:
|
||||
```json
|
||||
[
|
||||
{ "name": "Alice", "age": 30 },
|
||||
"Just a string",
|
||||
42,
|
||||
{ "name": "Charlie", "age": 35 }
|
||||
]
|
||||
```
|
||||
We may support heterogeneous arrays in the future with std::variant types.
|
||||
- We parse the first JSON document encountered in the string. Trailing
|
||||
characters are ignored. Thus if your JSON begins with {"a":1}, everything
|
||||
after the closing brace is ignored. This limitation will be lifted in the future,
|
||||
reporting an error.
|
||||
|
||||
These limitations are safe in the sense that they result in compile-time errors.
|
||||
Thus you will not get truncated strings or imprecise floats silently.
|
||||
|
||||
|
||||
Although we are committed to maintaining the functionality in the long run, the
|
||||
`compile_time::parse_json` function is subject to change.
|
||||
@@ -0,0 +1,445 @@
|
||||
# Compile-Time JSONPath and JSON Pointer Accessors
|
||||
|
||||
**Note:** This feature requires C++26 Static Reflection support (P2996) and is currently only available with experimental compilers. You must enable it with `-DSIMDJSON_STATIC_REFLECTION=ON` when building.
|
||||
|
||||
## Overview
|
||||
|
||||
simdjson provides compile-time JSONPath and JSON Pointer accessors that validate paths against struct definitions at compile time and generate optimized accessor code with zero runtime overhead. This combines the safety of compile-time type checking with the performance of pre-parsed, pre-validated access paths.
|
||||
|
||||
## Requirements
|
||||
|
||||
- C++26 compiler with Static Reflection support (P2996)
|
||||
- Experimental compiler flags:
|
||||
- Clang with P2996 support: `-std=c++26 -freflection -fexpansion-statements`
|
||||
- Build configuration: `-DSIMDJSON_STATIC_REFLECTION=ON`
|
||||
|
||||
## How It Works
|
||||
|
||||
**Compile Time:**
|
||||
1. Path string is parsed and converted to access steps
|
||||
2. Path is validated against struct definition using reflection
|
||||
3. Field types are checked and verified
|
||||
4. Optimized accessor code is generated
|
||||
|
||||
**Runtime:**
|
||||
- Direct navigation with no parsing
|
||||
- No validation overhead
|
||||
- No string comparisons for path components
|
||||
- Type-safe extraction
|
||||
|
||||
## Two Usage Modes
|
||||
|
||||
### Mode 1: With Type Validation (Recommended)
|
||||
|
||||
When you provide a struct type, the compiler validates the entire path at compile time:
|
||||
|
||||
```cpp
|
||||
struct User {
|
||||
std::string name;
|
||||
int age;
|
||||
std::vector<std::string> emails;
|
||||
};
|
||||
|
||||
// R"( ... )" is a C++ raw string literal.
|
||||
const padded_string json = R"({
|
||||
"name": "Alice",
|
||||
"age": 30,
|
||||
"emails": ["alice@example.com", "alice@work.com"]
|
||||
})"_padded;
|
||||
|
||||
ondemand::parser parser;
|
||||
auto doc = parser.iterate(json);
|
||||
|
||||
// Compile-time validation: checks that User has "name" field of type std::string
|
||||
std::string name;
|
||||
auto result = ondemand::json_path::at_path_compiled<User, ".name">(doc);
|
||||
result.get(name); // name = "Alice"
|
||||
|
||||
// Compile-time validation: checks that "emails" is array-like with string elements
|
||||
std::string email;
|
||||
result = ondemand::json_path::at_path_compiled<User, ".emails[0]">(doc);
|
||||
result.get(email); // email = "alice@example.com"
|
||||
```
|
||||
|
||||
**Benefits:**
|
||||
- **Compile-time errors** if path doesn't exist in struct
|
||||
- **Type safety** - verifies field types match expected types
|
||||
- **Refactoring protection** - renaming struct fields causes compile errors
|
||||
|
||||
**What gets validated:**
|
||||
- Field existence
|
||||
- Field types
|
||||
- Array/container access validity
|
||||
- Nested struct navigation
|
||||
|
||||
### Mode 2: Without Validation
|
||||
|
||||
When you omit the struct type, the path is parsed at compile time but not validated:
|
||||
|
||||
```cpp
|
||||
const padded_string json = R"({
|
||||
"name": "Alice",
|
||||
"age": 30,
|
||||
"address": {"city": "Boston"}
|
||||
})"_padded;
|
||||
|
||||
ondemand::parser parser;
|
||||
auto doc = parser.iterate(json);
|
||||
|
||||
// No compile-time validation - path is only parsed
|
||||
std::string name;
|
||||
auto result = ondemand::json_path::at_path_compiled<".name">(doc);
|
||||
result.get(name); // name = "Alice"
|
||||
|
||||
std::string_view city;
|
||||
result = ondemand::json_path::at_path_compiled<".address.city">(doc);
|
||||
result.get(city); // city = "Boston"
|
||||
```
|
||||
|
||||
**Benefits:**
|
||||
- Works with dynamic/unknown JSON structures
|
||||
- Still benefits from compile-time path parsing
|
||||
- No runtime string parsing overhead
|
||||
|
||||
**Use when:**
|
||||
- JSON structure is not known at compile time
|
||||
- Working with varied JSON schemas
|
||||
- Prototyping or exploratory parsing
|
||||
|
||||
## JSONPath Syntax
|
||||
|
||||
JSONPath uses dot notation and bracket notation for field access:
|
||||
|
||||
### Supported Syntax
|
||||
|
||||
| Syntax | Description | Example |
|
||||
|--------|-------------|---------|
|
||||
| `.field` | Dot notation for field access | `.name`, `.address.city` |
|
||||
| `["field"]` | Bracket notation with quotes | `["name"]`, `["address"]["city"]` |
|
||||
| `[index]` | Array index access | `[0]`, `[1]` |
|
||||
| Mixed | Combination of notations | `.emails[0]`, `["users"][0].name` |
|
||||
| `$` prefix | Optional root indicator | `$.name`, `$["name"]` |
|
||||
|
||||
### Examples
|
||||
|
||||
```cpp
|
||||
struct Address {
|
||||
std::string city;
|
||||
int zip;
|
||||
};
|
||||
|
||||
struct Person {
|
||||
std::string name;
|
||||
int age;
|
||||
Address address;
|
||||
std::vector<std::string> emails;
|
||||
};
|
||||
|
||||
// Dot notation
|
||||
at_path_compiled<Person, ".name">(doc)
|
||||
at_path_compiled<Person, ".address.city">(doc)
|
||||
|
||||
// Bracket notation
|
||||
at_path_compiled<Person, "[\"name\"]">(doc)
|
||||
at_path_compiled<Person, "[\"address\"][\"city\"]">(doc)
|
||||
|
||||
// Array access
|
||||
at_path_compiled<Person, ".emails[0]">(doc)
|
||||
at_path_compiled<Person, ".emails[1]">(doc)
|
||||
|
||||
// Mixed notation
|
||||
at_path_compiled<Person, ".address[\"zip\"]">(doc)
|
||||
at_path_compiled<Person, "[\"emails\"][0]">(doc)
|
||||
|
||||
// With root indicator
|
||||
at_path_compiled<Person, "$.name">(doc)
|
||||
at_path_compiled<Person, "$.address.city">(doc)
|
||||
```
|
||||
|
||||
## JSON Pointer Syntax
|
||||
|
||||
JSON Pointer (RFC 6901) uses slash-separated paths:
|
||||
|
||||
### Supported Syntax
|
||||
|
||||
| Syntax | Description | Example |
|
||||
|--------|-------------|---------|
|
||||
| `/field` | Field access | `/name`, `/address/city` |
|
||||
| `/index` | Array index | `/0`, `/1` |
|
||||
| `~0` | Escaped `~` | `/field~0name` → field~name |
|
||||
| `~1` | Escaped `/` | `/field~1name` → field/name |
|
||||
|
||||
### Examples
|
||||
|
||||
```cpp
|
||||
struct Car {
|
||||
std::string make;
|
||||
std::string model;
|
||||
int64_t year;
|
||||
std::vector<double> tire_pressure;
|
||||
};
|
||||
|
||||
// Field access
|
||||
at_pointer_compiled<Car, "/make">(doc)
|
||||
at_pointer_compiled<Car, "/model">(doc)
|
||||
|
||||
// Array access
|
||||
at_pointer_compiled<Car, "/tire_pressure/0">(doc)
|
||||
at_pointer_compiled<Car, "/tire_pressure/1">(doc)
|
||||
|
||||
// Root pointer (returns whole document)
|
||||
at_pointer_compiled<Car, "">(doc)
|
||||
at_pointer_compiled<Car, "/">(doc)
|
||||
```
|
||||
|
||||
## API Reference
|
||||
|
||||
### JSONPath Functions
|
||||
|
||||
```cpp
|
||||
// With type validation
|
||||
template<typename T, constevalutil::fixed_string Path, typename DocOrValue>
|
||||
simdjson_result<value> at_path_compiled(DocOrValue& doc_or_val);
|
||||
|
||||
// Without validation
|
||||
template<constevalutil::fixed_string Path, typename DocOrValue>
|
||||
simdjson_result<value> at_path_compiled(DocOrValue& doc_or_val);
|
||||
```
|
||||
|
||||
### JSON Pointer Functions
|
||||
|
||||
```cpp
|
||||
// With type validation
|
||||
template<typename T, constevalutil::fixed_string Pointer, typename DocOrValue>
|
||||
simdjson_result<value> at_pointer_compiled(DocOrValue& doc_or_val);
|
||||
|
||||
// Without validation
|
||||
template<constevalutil::fixed_string Pointer, typename DocOrValue>
|
||||
simdjson_result<value> at_pointer_compiled(DocOrValue& doc_or_val);
|
||||
```
|
||||
|
||||
### Direct Field Extraction
|
||||
|
||||
Extract values directly into variables with compile-time type checking:
|
||||
|
||||
```cpp
|
||||
// JSONPath
|
||||
template<typename T, constevalutil::fixed_string Path>
|
||||
struct path_accessor {
|
||||
template<typename DocOrValue, typename FieldType>
|
||||
static error_code extract_field(DocOrValue& doc_or_val, FieldType& target);
|
||||
};
|
||||
|
||||
// JSON Pointer
|
||||
template<typename T, constevalutil::fixed_string Pointer>
|
||||
struct pointer_accessor {
|
||||
template<typename DocOrValue, typename FieldType>
|
||||
static error_code extract_field(DocOrValue& doc_or_val, FieldType& target);
|
||||
};
|
||||
```
|
||||
|
||||
**Example:**
|
||||
|
||||
```cpp
|
||||
struct User {
|
||||
std::string name;
|
||||
int age;
|
||||
};
|
||||
|
||||
ondemand::parser parser;
|
||||
auto doc = parser.iterate(json);
|
||||
|
||||
// Extract directly into variable
|
||||
std::string name;
|
||||
ondemand::json_path::path_accessor<User, ".name">::extract_field(doc, name);
|
||||
|
||||
int age;
|
||||
ondemand::json_path::pointer_accessor<User, "/age">::extract_field(doc, age);
|
||||
```
|
||||
|
||||
The compiler verifies that the target variable type matches the field type at the path.
|
||||
|
||||
## Complete Examples
|
||||
|
||||
### Example 1: Validated Access
|
||||
|
||||
```cpp
|
||||
#include "simdjson.h"
|
||||
using namespace simdjson;
|
||||
|
||||
struct Car {
|
||||
std::string make;
|
||||
std::string model;
|
||||
int64_t year;
|
||||
std::vector<double> tire_pressure;
|
||||
};
|
||||
|
||||
int main() {
|
||||
const padded_string json = R"({
|
||||
"make": "Toyota",
|
||||
"model": "Camry",
|
||||
"year": 2018,
|
||||
"tire_pressure": [40.1, 39.9, 37.7, 40.4]
|
||||
})"_padded;
|
||||
|
||||
ondemand::parser parser;
|
||||
auto doc = parser.iterate(json);
|
||||
|
||||
// Type-validated access
|
||||
std::string make;
|
||||
auto result = ondemand::json_path::at_path_compiled<Car, ".make">(doc);
|
||||
result.get(make); // make = "Toyota"
|
||||
|
||||
// Array access with validation
|
||||
double pressure;
|
||||
result = ondemand::json_path::at_path_compiled<Car, ".tire_pressure[1]">(doc);
|
||||
result.get(pressure); // pressure = 39.9
|
||||
|
||||
return 0;
|
||||
}
|
||||
```
|
||||
|
||||
### Example 2: Non-Validated Access
|
||||
|
||||
```cpp
|
||||
#include "simdjson.h"
|
||||
using namespace simdjson;
|
||||
|
||||
int main() {
|
||||
const padded_string json = R"({
|
||||
"user": {
|
||||
"name": "Alice",
|
||||
"preferences": {
|
||||
"theme": "dark",
|
||||
"notifications": true
|
||||
}
|
||||
}
|
||||
})"_padded;
|
||||
|
||||
ondemand::parser parser;
|
||||
auto doc = parser.iterate(json);
|
||||
|
||||
// No validation - works with any JSON structure
|
||||
std::string_view theme;
|
||||
auto result = ondemand::json_path::at_path_compiled<".user.preferences.theme">(doc);
|
||||
result.get(theme); // theme = "dark"
|
||||
|
||||
bool notifications;
|
||||
result = ondemand::json_path::at_path_compiled<".user.preferences.notifications">(doc);
|
||||
result.get(notifications); // notifications = true
|
||||
|
||||
return 0;
|
||||
}
|
||||
```
|
||||
|
||||
### Example 3: Direct Extraction
|
||||
|
||||
```cpp
|
||||
#include "simdjson.h"
|
||||
using namespace simdjson;
|
||||
|
||||
struct Person {
|
||||
std::string name;
|
||||
int age;
|
||||
std::vector<std::string> emails;
|
||||
};
|
||||
|
||||
int main() {
|
||||
const padded_string json = R"({
|
||||
"name": "Bob",
|
||||
"age": 25,
|
||||
"emails": ["bob@example.com", "bob@work.com"]
|
||||
})"_padded;
|
||||
|
||||
ondemand::parser parser;
|
||||
auto doc = parser.iterate(json);
|
||||
|
||||
// Extract with type validation
|
||||
std::string name;
|
||||
ondemand::json_path::path_accessor<Person, ".name">::extract_field(doc, name);
|
||||
// name = "Bob"
|
||||
|
||||
int age;
|
||||
ondemand::json_path::pointer_accessor<Person, "/age">::extract_field(doc, age);
|
||||
// age = 25
|
||||
|
||||
std::string email;
|
||||
ondemand::json_path::path_accessor<Person, ".emails[0]">::extract_field(doc, email);
|
||||
// email = "bob@example.com"
|
||||
|
||||
return 0;
|
||||
}
|
||||
```
|
||||
|
||||
## Error Handling
|
||||
|
||||
Compile-time errors occur when:
|
||||
- Path doesn't exist in struct: `static_assert` failure
|
||||
- Field type mismatch: `static_assert` failure
|
||||
- Invalid array access on non-array field: `static_assert` failure
|
||||
|
||||
Runtime errors occur when:
|
||||
- JSON structure doesn't match expected structure
|
||||
- Array index out of bounds
|
||||
- Type conversion failures
|
||||
|
||||
```cpp
|
||||
struct User {
|
||||
std::string name;
|
||||
int age;
|
||||
};
|
||||
|
||||
// Compile-time error: no "email" field in User
|
||||
// auto result = ondemand::json_path::at_path_compiled<User, ".email">(doc);
|
||||
|
||||
// Compile-time error: age is not an array
|
||||
// auto result = ondemand::json_path::at_path_compiled<User, ".age[0]">(doc);
|
||||
|
||||
// Runtime error if JSON doesn't have "name" field
|
||||
auto result = ondemand::json_path::at_path_compiled<User, ".name">(doc);
|
||||
std::string name;
|
||||
if (result.get(name) != SUCCESS) {
|
||||
// Handle error
|
||||
}
|
||||
```
|
||||
|
||||
## Performance
|
||||
|
||||
Compile-time accessors provide:
|
||||
- **Zero path parsing overhead** - paths parsed at compile time
|
||||
- **Zero validation overhead** - validation done at compile time
|
||||
- **Direct field access** - no runtime path traversal
|
||||
- **Type-safe extraction** - no dynamic type checking
|
||||
|
||||
Compared to runtime `at_path()` and `at_pointer()`:
|
||||
- Eliminates runtime path string parsing
|
||||
- Eliminates runtime path validation
|
||||
- Generates optimal code path directly
|
||||
|
||||
## Limitations
|
||||
|
||||
- Requires C++26 compiler with P2996 support (experimental)
|
||||
- Paths must be compile-time constants (string literals)
|
||||
- Cannot use runtime-computed paths
|
||||
- Limited to struct types that support reflection
|
||||
- Array indices must be compile-time constants in the path
|
||||
|
||||
## When to Use
|
||||
|
||||
**Use compile-time accessors when:**
|
||||
- You have well-defined struct types
|
||||
- JSON structure is known at compile time
|
||||
- You want maximum type safety
|
||||
- Performance is critical
|
||||
|
||||
**Use runtime `at_path()`/`at_pointer()` when:**
|
||||
- JSON structure varies or is unknown
|
||||
- Paths are computed at runtime
|
||||
- Working with C++20 or earlier
|
||||
- Flexibility is more important than compile-time checks
|
||||
|
||||
## See Also
|
||||
|
||||
- [JSON Pointer](basics.md#json-pointer) - Runtime JSON Pointer support
|
||||
- [JSONPath](basics.md#jsonpath) - Runtime JSONPath support
|
||||
- [Static Reflection for Deserialization](basics.md#3-using-static-reflection-c26) - Using reflection for full struct deserialization
|
||||
@@ -41,7 +41,7 @@ The Basics: Loading and Parsing JSON Documents using the DOM front-end
|
||||
The simdjson library offers a simple DOM tree API, which you can access by creating a
|
||||
`dom::parser` and calling the `load()` method:
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
dom::parser parser;
|
||||
dom::element doc = parser.load(filename); // load and parse a file
|
||||
```
|
||||
@@ -49,21 +49,39 @@ dom::element doc = parser.load(filename); // load and parse a file
|
||||
Or by creating a padded string (for efficiency reasons, simdjson requires a string with
|
||||
SIMDJSON_PADDING bytes at the end) and calling `parse()`:
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
dom::parser parser;
|
||||
dom::element doc = parser.parse("[1,2,3]"_padded); // parse a string, the _padded suffix creates a simdjson::padded_string instance
|
||||
```
|
||||
|
||||
You can also load a `padded_string` from a file.
|
||||
|
||||
|
||||
```cpp
|
||||
auto json = padded_string::load("twitter.json"); // load JSON file 'twitter.json'.
|
||||
dom::element doc = parser.parse(json);
|
||||
```
|
||||
|
||||
[You can similarly fetch a file from a URL to a padded string](https://github.com/simdjson/curltostring) using our `simdjson::padded_string_builder`.
|
||||
|
||||
(Windows users compiling with C++17 or better may use `wchar_t` strings to support non-ASCII
|
||||
filenames: `padded_string::load(L"twitter.json")`.)
|
||||
|
||||
|
||||
(Windows users compiling with C++17 or better may use `wchar_t` strings to support non-ASCII
|
||||
filenames: `padded_string::load(L"twitter.json")`.)
|
||||
|
||||
|
||||
You can copy your data directly on a `simdjson::padded_string` as follows:
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
const char * data = "my data"; // 7 bytes
|
||||
simdjson::padded_string my_padded_data(data, 7); // copies to a padded buffer
|
||||
```
|
||||
|
||||
Or as follows...
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
std::string data = "my data";
|
||||
simdjson::padded_string my_padded_data(data); // copies to a padded buffer
|
||||
```
|
||||
@@ -83,7 +101,7 @@ container-overflow checks, you may encounter sanitizer warnings.
|
||||
You can safely ignore these warnings. Or you can call `simdjson::pad(std::string&)` to pad the
|
||||
string with `SIMDJSON_PADDING` spaces: this function returns a `simdjson::padding_string_view` which can be be passed to the parser's iterator function:
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
std::string json = "[1]";
|
||||
dom::element doc = parser.parse(simdjson::pad(json));
|
||||
```
|
||||
@@ -117,8 +135,9 @@ Once you have an element, you can navigate it with idiomatic C++ iterators, oper
|
||||
dom::object and dom::array. An exception (`simdjson::simdjson_error`) is thrown if the cast is not possible.
|
||||
* **Extracting Values (without exceptions):** You can use a variant usage of `get()` with error codes to avoid exceptions. You first declare the variable of the appropriate type (`double`, `uint64_t`, `int64_t`, `bool`, `std::string_view`,
|
||||
`dom::object` and `dom::array`) and pass it by reference to `get()` which gives you back an error code: e.g.,
|
||||
```c++
|
||||
```cpp
|
||||
simdjson::error_code error;
|
||||
// _padded returns an simdjson::padded_string instance
|
||||
simdjson::padded_string numberstring = "1.2"_padded; // our JSON input ("1.2")
|
||||
simdjson::dom::parser parser;
|
||||
double value; // variable where we store the value to be parsed
|
||||
@@ -152,7 +171,8 @@ Once you have an element, you can navigate it with idiomatic C++ iterators, oper
|
||||
|
||||
The following code illustrates all of the above:
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
// R"( ... )" is a C++ raw string literal.
|
||||
auto cars_json = R"( [
|
||||
{ "make": "Toyota", "model": "Camry", "year": 2018, "tire_pressure": [ 40.1, 39.9, 37.7, 40.4 ] },
|
||||
{ "make": "Kia", "model": "Soul", "year": 2012, "tire_pressure": [ 30.1, 31.0, 28.6, 28.7 ] },
|
||||
@@ -185,7 +205,7 @@ for (dom::object car : parser.parse(cars_json)) {
|
||||
|
||||
Here is a different example illustrating the same ideas:
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
auto abstract_json = R"( [
|
||||
{ "12345" : {"a":12.34, "b":56.78, "c": 9998877} },
|
||||
{ "12545" : {"a":11.44, "b":12.78, "c": 11111111} }
|
||||
@@ -207,7 +227,7 @@ for (dom::object obj : parser.parse(abstract_json)) {
|
||||
And another one:
|
||||
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
auto abstract_json = R"(
|
||||
{ "str" : { "123" : {"abc" : 3.14 } } } )"_padded;
|
||||
dom::parser parser;
|
||||
@@ -221,7 +241,7 @@ C++17 Support
|
||||
|
||||
While the simdjson library can be used in any project using C++ 11 and above, field iteration has special support C++ 17's destructuring syntax. For example:
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
padded_string json = R"( { "foo": 1, "bar": 2 } )"_padded;
|
||||
dom::parser parser;
|
||||
dom::object object; // invalid until the get() succeeds
|
||||
@@ -234,7 +254,7 @@ for (auto [key, value] : object) {
|
||||
|
||||
For comparison, here is the C++ 11 version of the same code:
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
// C++ 11 version for comparison
|
||||
padded_string json = R"( { "foo": 1, "bar": 2 } )"_padded;
|
||||
dom::parser parser;
|
||||
@@ -251,7 +271,7 @@ C++20 Support
|
||||
|
||||
simdjson library also supports some C++20 feature including `std::ranges`:
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
auto cars_json = R"( [
|
||||
{ "make": "Toyota", "model": "Camry", "year": 2018, "tire_pressure": [ 40.1, 39.9, 37.7, 40.4 ] },
|
||||
{ "make": "Kia", "model": "Soul", "year": 2012, "tire_pressure": [ 30.1, 31.0, 28.6, 28.7 ] },
|
||||
@@ -270,7 +290,7 @@ JSON Pointer
|
||||
The simdjson library also supports [JSON pointer](https://tools.ietf.org/html/rfc6901) through the
|
||||
`at_pointer()` method, letting you reach further down into the document in a single call:
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
auto cars_json = R"( [
|
||||
{ "make": "Toyota", "model": "Camry", "year": 2018, "tire_pressure": [ 40.1, 39.9, 37.7, 40.4 ] },
|
||||
{ "make": "Kia", "model": "Soul", "year": 2012, "tire_pressure": [ 30.1, 31.0, 28.6, 28.7 ] },
|
||||
@@ -291,7 +311,7 @@ You can apply a JSON Pointer expression to any node and the path gets interprete
|
||||
|
||||
Consider the following example:
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
auto cars_json = R"( [
|
||||
{ "make": "Toyota", "model": "Camry", "year": 2018, "tire_pressure": [ 40.1, 39.9, 37.7, 40.4 ] },
|
||||
{ "make": "Kia", "model": "Soul", "year": 2012, "tire_pressure": [ 30.1, 31.0, 28.6, 28.7 ] },
|
||||
@@ -313,11 +333,11 @@ JSONPath
|
||||
------------
|
||||
|
||||
|
||||
The simdjson library supports a subset of [JSONPath](https://datatracker.ietf.org/doc/html/draft-normington-jsonpath-00) through the `at_path()` method, allowing you to reach further into the document in a single call. The subset of JSONPath that is implemented is the subset that is trivially convertible into the JSON Pointer format, using `.` to access a field and `[]` to access a specific index.
|
||||
The simdjson library supports a subset of [JSONPath](https://www.rfc-editor.org/rfc/rfc9535) (RFC 9535) through the `at_path()` method, allowing you to reach further into the document in a single call. The subset of JSONPath that is implemented is the subset that is trivially convertible into the JSON Pointer format, using `.` to access a field and `[]` to access a specific index.
|
||||
|
||||
Consider the following example:
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
auto cars_json = R"( [
|
||||
{ "make": "Toyota", "model": "Camry", "year": 2018, "tire_pressure": [ 40.1, 39.9, 37.7, 40.4 ] },
|
||||
{ "make": "Kia", "model": "Soul", "year": 2012, "tire_pressure": [ 30.1, 31.0, 28.6, 28.7 ] },
|
||||
@@ -336,7 +356,7 @@ cout << p << endl; // Prints 39.9
|
||||
|
||||
We also support the `$` prefix. When you start a JSONPath expression with $, you are indicating that the path starts from the root of the JSON document. E.g.,
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
auto json = R"( { "c" :{ "foo": { "a": [ 10, 20, 30 ] }}, "d": { "foo2": { "a": [ 10, 20, 30 ] }} , "e": 120 })"_padded;
|
||||
dom::parser parser;
|
||||
dom::element doc;
|
||||
@@ -428,7 +448,7 @@ Error Handling
|
||||
All simdjson APIs that can fail return `simdjson_result<T>`, which is a <value, error_code>
|
||||
pair. You can retrieve the value with .get(), like so:
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
dom::element doc;
|
||||
auto error = parser.parse(json).get(doc);
|
||||
if (error) { cerr << error << endl; exit(1); }
|
||||
@@ -462,7 +482,7 @@ We can write a "quick start" example where we attempt to parse the following JSO
|
||||
Our program loads the file, selects value corresponding to key "search_metadata" which expected to be an object, and then
|
||||
it selects the key "count" within that object.
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
#include <iostream>
|
||||
#include "simdjson.h"
|
||||
|
||||
@@ -490,7 +510,7 @@ triggering exceptions. To do this, we use `["statuses"].at(0)["id"]`. We break t
|
||||
|
||||
Observe how we use the `at` method when querying an index into an array, and not the bracket operator.
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
#include <iostream>
|
||||
#include "simdjson.h"
|
||||
|
||||
@@ -514,7 +534,7 @@ over the content of an array.
|
||||
|
||||
This is how the example in "Using the Parsed JSON" could be written using only error code checking:
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
auto cars_json = R"( [
|
||||
{ "make": "Toyota", "model": "Camry", "year": 2018, "tire_pressure": [ 40.1, 39.9, 37.7, 40.4 ] },
|
||||
{ "make": "Kia", "model": "Soul", "year": 2012, "tire_pressure": [ 30.1, 31.0, 28.6, 28.7 ] },
|
||||
@@ -561,7 +581,7 @@ for (dom::element car_element : cars) {
|
||||
|
||||
Here is another example:
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
auto abstract_json = R"( [
|
||||
{ "12345" : {"a":12.34, "b":56.78, "c": 9998877} },
|
||||
{ "12545" : {"a":11.44, "b":12.78, "c": 11111111} }
|
||||
@@ -594,7 +614,7 @@ for (dom::element elem : array) {
|
||||
|
||||
And another one:
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
auto abstract_json = R"(
|
||||
{ "str" : { "123" : {"abc" : 3.14 } } } )"_padded;
|
||||
dom::parser parser;
|
||||
@@ -608,7 +628,7 @@ Notice how we can string several operations (`parser.parse(abstract_json)["str"]
|
||||
|
||||
The next two functions will take as input a JSON document containing an array with a single element, either a string or a number. They return true upon success.
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
simdjson::dom::parser parser{};
|
||||
|
||||
bool parse_double(const char *j, double &d) {
|
||||
@@ -640,7 +660,7 @@ target_compile_definitions(simdjson PUBLIC SIMDJSON_EXCEPTIONS=OFF)
|
||||
|
||||
Users more comfortable with an exception flow may choose to directly cast the `simdjson_result<T>` to the desired type:
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
dom::element doc = parser.parse(json); // Throws an exception if there was an error!
|
||||
```
|
||||
|
||||
@@ -650,7 +670,7 @@ program from continuing if there was an error.
|
||||
|
||||
If one is willing to trigger exceptions, it is possible to write simpler code:
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
#include <iostream>
|
||||
#include "simdjson.h"
|
||||
|
||||
@@ -671,7 +691,7 @@ inspect or walk over JSON elements. To do that, you can use iterators and the ty
|
||||
example, here's a quick and dirty recursive function that verbosely prints the JSON document as JSON
|
||||
(* ignoring nuances like trailing commas and escaping strings, for brevity's sake):
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
void print_json(dom::element element) {
|
||||
switch (element.type()) {
|
||||
case dom::element_type::ARRAY:
|
||||
@@ -727,7 +747,7 @@ and reuse it. The simdjson library will allocate and retain internal buffers bet
|
||||
buffers hot in cache and keeping memory allocation and initialization to a minimum. In this manner,
|
||||
you can parse terabytes of JSON data without doing any new allocation.
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
dom::parser parser;
|
||||
|
||||
// This initializes buffers and a document big enough to handle this JSON.
|
||||
@@ -770,7 +790,7 @@ without bound:
|
||||
|
||||
* You can set a *max capacity* when constructing a parser:
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
dom::parser parser(1000*1000); // Never grow past documents > 1MB
|
||||
for (web_request request : listen()) {
|
||||
dom::element doc;
|
||||
@@ -786,7 +806,7 @@ without bound:
|
||||
* You can set a *fixed capacity* that never grows, as well, which can be excellent for
|
||||
predictability and reliability, since simdjson will never call malloc after startup!
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
dom::parser parser(0); // This parser will refuse to automatically grow capacity
|
||||
auto error = parser.allocate(1000*1000); // This allocates enough capacity to handle documents <= 1MB
|
||||
if (error) { cerr << error << endl; exit(1); }
|
||||
@@ -817,7 +837,7 @@ When calling `parser.parse` on a pointer (e.g., `parser.parse(my_char_pointer, m
|
||||
Some users may not be able use our `padded_string` class or to load the data directly from disk (`parser.load`). They may need to pass data pointers to the library. If these users wish to avoid temporary copies and corresponding temporary memory allocations, they may want to call `parser.parse` with the `realloc_if_needed` parameter set to false (e.g., `parser.parse(my_char_pointer, my_length_in_bytes, false)`). In such cases, they need to ensure that there are at least SIMDJSON_PADDING extra bytes at the end that can be safely accessed and read. They do not need to initialize the padded bytes to any value in particular. The following example is safe:
|
||||
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
const char *json = R"({"key":"value"})";
|
||||
const size_t json_len = std::strlen(json);
|
||||
std::unique_ptr<char[]> padded_json_copy{new char[json_len + SIMDJSON_PADDING]};
|
||||
@@ -825,7 +845,7 @@ memcpy(padded_json_copy.get(), json, json_len);
|
||||
memset(padded_json_copy.get() + json_len, 0, SIMDJSON_PADDING);
|
||||
simdjson::dom::parser parser;
|
||||
simdjson::dom::element element = parser.parse(padded_json_copy.get(), json_len, false);
|
||||
````
|
||||
```
|
||||
|
||||
Setting the `realloc_if_needed` parameter `false` in this manner may lead to better performance since copies are avoided, but it requires that the user takes more responsibilities: the simdjson library cannot verify that the input buffer was padded with SIMDJSON_PADDING extra bytes.
|
||||
|
||||
|
||||
@@ -55,7 +55,7 @@ Inspecting the Detected Implementation
|
||||
|
||||
You can check what implementation is running with `active_implementation`:
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
cout << "simdjson v" << SIMDJSON_VERSION << endl;
|
||||
cout << "Detected the best implementation for your machine: " << simdjson::get_active_implementation()->name();
|
||||
cout << "(" << simdjson::get_active_implementation()->description() << ")" << endl;
|
||||
@@ -68,7 +68,7 @@ Querying Available Implementations
|
||||
|
||||
You can list all available implementations, regardless of which one was selected:
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
for (auto implementation : simdjson::get_available_implementations()) {
|
||||
cout << implementation->name() << ": " << implementation->description() << endl;
|
||||
}
|
||||
@@ -76,7 +76,7 @@ for (auto implementation : simdjson::get_available_implementations()) {
|
||||
|
||||
And look them up by name:
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
cout << simdjson::get_available_implementations()["fallback"]->description() << endl;
|
||||
```
|
||||
When an implementation is not available, the bracket call `simdjson::get_available_implementations()[name]`
|
||||
@@ -93,7 +93,7 @@ Manually Selecting the Implementation
|
||||
If you're trying to do performance tests or see how different implementations of simdjson run, you
|
||||
can select the CPU architecture yourself:
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
// Use the fallback implementation, even though my machine is fast enough for anything
|
||||
simdjson::get_active_implementation() = simdjson::get_available_implementations()["fallback"];
|
||||
```
|
||||
@@ -102,7 +102,7 @@ You are responsible for ensuring that the requirements of the selected implement
|
||||
Furthermore, you should check that the implementation is available before setting it to `simdjson::get_active_implementation()`
|
||||
by comparing it with the null pointer.
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
auto my_implementation = simdjson::get_available_implementations()["haswell"];
|
||||
if (! my_implementation) { exit(1); }
|
||||
if (! my_implementation->supported_by_runtime_system()) { exit(1); }
|
||||
@@ -114,7 +114,7 @@ Checking that an Implementation can Run on your System
|
||||
|
||||
You should call `supported_by_runtime_system()` to compare the processor's features with the need of the implementation.
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
for (auto implementation : simdjson::get_available_implementations()) {
|
||||
if (implementation->supported_by_runtime_system()) {
|
||||
cout << implementation->name() << ": " << implementation->description() << endl;
|
||||
|
||||
@@ -8,7 +8,7 @@ library provides high-speed access to files or streams containing multiple small
|
||||
{"text":"a"}
|
||||
{"text":"b"}
|
||||
{"text":"c"}
|
||||
...
|
||||
"..."
|
||||
```
|
||||
... you want to read the entries (individual JSON documents) as quickly and as conveniently as possible. Importantly, the input might span several gigabytes, but you want to use a small (fixed) amount of memory. Ideally, you'd also like the parallelize the processing (using more than one core) to speed up the process.
|
||||
|
||||
@@ -132,7 +132,7 @@ E.g., `[1,2]{"32":1}` is recognized as two documents.
|
||||
Some official formats **(non-exhaustive list)**:
|
||||
- [Newline-Delimited JSON (NDJSON)](https://github.com/ndjson/ndjson-spec/)
|
||||
- [JSON lines (JSONL)](http://jsonlines.org/)
|
||||
- [Record separator-delimited JSON (RFC 7464)](https://tools.ietf.org/html/rfc7464) <- Not supported by JsonStream!
|
||||
- [Record separator-delimited JSON (RFC 7464)](https://tools.ietf.org/html/rfc7464) <- Not supported by simdjson!
|
||||
- [More on Wikipedia...](https://en.wikipedia.org/wiki/JSON_streaming)
|
||||
|
||||
API
|
||||
@@ -140,8 +140,10 @@ API
|
||||
|
||||
Example:
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
// R"( ... )" is a C++ raw string literal.
|
||||
auto json = R"({ "foo": 1 } { "foo": 2 } { "foo": 3 } )"_padded;
|
||||
// _padded returns an simdjson::padded_string instance
|
||||
ondemand::parser parser;
|
||||
ondemand::document_stream docs = parser.iterate_many(json);
|
||||
for (auto doc : docs) {
|
||||
@@ -197,7 +199,7 @@ and `error()` to check if there were any error.
|
||||
Let us illustrate the idea with code:
|
||||
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
auto json = R"([1,2,3] {"1":1,"2":3,"4":4} [1,2,3] )"_padded;
|
||||
simdjson::ondemand::parser parser;
|
||||
simdjson::ondemand::document_stream stream;
|
||||
@@ -238,7 +240,7 @@ Some users may need to work with truncated streams. The simdjson may truncate do
|
||||
|
||||
Consider the following example where a truncated document (`{"key":"intentionally unclosed string `) containing 39 bytes has been left within the stream. In such cases, the first two whole documents are parsed and returned, and the `truncated_bytes()` method returns 39.
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
auto json = R"([1,2,3] {"1":1,"2":3,"4":4} {"key":"intentionally unclosed string )"_padded;
|
||||
simdjson::ondemand::parser parser;
|
||||
simdjson::ondemand::document_stream stream;
|
||||
@@ -267,7 +269,7 @@ is effectively ignored, as it is set to at least the document size.
|
||||
|
||||
Example:
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
auto json = R"( 1, 2, 3, 4, "a", "b", "c", {"hello": "world"} , [1, 2, 3])"_padded;
|
||||
ondemand::parser parser;
|
||||
ondemand::document_stream doc_stream;
|
||||
@@ -314,7 +316,7 @@ the simdjson library.
|
||||
|
||||
Consider a custom class `Car`:
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
struct Car {
|
||||
std::string make;
|
||||
std::string model;
|
||||
@@ -328,7 +330,7 @@ You may support deserializing directly from a JSON value or document to your own
|
||||
by defining a single `tag_invoke` function:
|
||||
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
namespace simdjson {
|
||||
// This tag_invoke MUST be inside simdjson namespace
|
||||
template <typename simdjson_value>
|
||||
@@ -370,7 +372,7 @@ tag_invoke functions.
|
||||
Given a stream of JSON documents, you can add them to a data structure
|
||||
such as a `std::vector<Car>` like so if you support exceptions:
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
padded_string json =
|
||||
R"( { "make": "Toyota", "model": "Camry", "year": 2018,
|
||||
"tire_pressure": [ 40.1, 39.9 ] }
|
||||
@@ -391,7 +393,7 @@ such as a `std::vector<Car>` like so if you support exceptions:
|
||||
Otherwise you may use this longer version for explicit handling of errors:
|
||||
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
std::vector<Car> cars;
|
||||
for(auto doc : stream) {
|
||||
Car c;
|
||||
@@ -401,4 +403,4 @@ Otherwise you may use this longer version for explicit handling of errors:
|
||||
}
|
||||
cars.push_back(c);
|
||||
}
|
||||
```
|
||||
```
|
||||
|
||||
@@ -23,7 +23,7 @@ applications with a computation efficiency that is difficult to surpass.
|
||||
|
||||
A code example illustrates our API from a programmer's point of view:
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
ondemand::parser parser;
|
||||
auto doc = parser.iterate(json);
|
||||
for (auto tweet : doc["statuses"]) {
|
||||
@@ -109,7 +109,7 @@ The DOM approach was the only way to parse JSON documents up to version 0.6 of t
|
||||
Our DOM API looks similar to our On-Demand example, except
|
||||
it calls `parse` instead of `iterate`:
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
dom::parser parser;
|
||||
auto doc = parser.parse(json);
|
||||
for (auto tweet : doc["statuses"]) {
|
||||
@@ -157,7 +157,7 @@ examples. To make it short enough to use as an example at all, it has heavily re
|
||||
a part of the problem (does not get user.screen_name), it has bugs (it does not handle sub-objects
|
||||
in a tweet at all), and it uses a theoretical, simple event-based API that minimizes ceremony.
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
struct twitter_callbacks {
|
||||
bool in_statuses;
|
||||
bool in_tweet;
|
||||
@@ -284,14 +284,14 @@ To help visualize the algorithm, we'll walk through the example C++ given at the
|
||||
|
||||
This declaration does not allocate any memory; that will happen in the next step.
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
ondemand::parser parser;
|
||||
```
|
||||
|
||||
2. We then start iterating the JSON document by allocating internal parser buffers, preprocessing
|
||||
the JSON, and initializing the iterator.
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
auto doc = parser.iterate(json);
|
||||
```
|
||||
|
||||
@@ -337,14 +337,14 @@ To help visualize the algorithm, we'll walk through the example C++ given at the
|
||||
3. We iterate over the "statuses" field using a typical C++ iterator, reading past the initial
|
||||
`{ "statuses": [ {`.
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
for (ondemand::object tweet : doc["statuses"]) {
|
||||
```
|
||||
|
||||
This shorthand does a lot, and it is helpful to see what it expands to.
|
||||
Comments in front of each one explain what's going on:
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
// Validate that the top-level value is an object: check for {. Increase depth to 2 (root > field).
|
||||
ondemand::object top = doc.get_object();
|
||||
|
||||
@@ -396,7 +396,7 @@ To help visualize the algorithm, we'll walk through the example C++ given at the
|
||||
|
||||
4. We get the `"text"` field as a string.
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
std::string_view text = tweet["text"];
|
||||
```
|
||||
|
||||
@@ -435,7 +435,7 @@ To help visualize the algorithm, we'll walk through the example C++ given at the
|
||||
|
||||
4. We get the `"screen_name"` from the `"user"` object.
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
ondemand::object user = tweet["user"];
|
||||
screen_name = user["screen_name"];
|
||||
```
|
||||
@@ -469,7 +469,7 @@ To help visualize the algorithm, we'll walk through the example C++ given at the
|
||||
|
||||
5. We get `"retweet_count"` as an unsigned integer.
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
uint64_t retweets = tweet["retweet_count"];
|
||||
```
|
||||
|
||||
@@ -513,7 +513,7 @@ To help visualize the algorithm, we'll walk through the example C++ given at the
|
||||
|
||||
6. We loop to the next tweet.
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
for (ondemand::object tweet : doc["statuses"]) {
|
||||
...
|
||||
}
|
||||
@@ -521,7 +521,7 @@ To help visualize the algorithm, we'll walk through the example C++ given at the
|
||||
|
||||
The relevant parts of the loop are:
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
while (iter != statuses.end()) {
|
||||
ondemand::object tweet = *iter;
|
||||
...
|
||||
@@ -545,7 +545,7 @@ To help visualize the algorithm, we'll walk through the example C++ given at the
|
||||
"statuses": [
|
||||
{ "id": 1, "text": "first!", "user": { "screen_name": "lemire", "name": "Daniel" }, "retweet_count": 40 },
|
||||
{ "id": 2, "text": "second!", "user": { "screen_name": "jkeiser2", "name": "John" }, "retweet_count": 3 }
|
||||
^ (depth 3 - root > statuses > tweet)
|
||||
^ (depth 4 - root > statuses > tweet > field)
|
||||
],
|
||||
"search_metadata": { "count": 2 }
|
||||
}
|
||||
@@ -566,7 +566,7 @@ To help visualize the algorithm, we'll walk through the example C++ given at the
|
||||
|
||||
8. The loop ends. Recall the relevant parts of the statuses loop:
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
while (iter != statuses.end()) {
|
||||
ondemand::object tweet = *iter;
|
||||
...
|
||||
@@ -610,7 +610,7 @@ When the user requests strings, we unescape them to a single string buffer much
|
||||
so that users enjoy the same string performance as the core simdjson. We do not write the length to the
|
||||
string buffer, however; that is stored in the `string_view` instance we return to the user.
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
ondemand::parser parser;
|
||||
auto doc = parser.iterate(json);
|
||||
std::set<std::string_view> default_users;
|
||||
@@ -645,7 +645,7 @@ from the `unescaped_key()` method has a lifecycle tied to the `parser` instance:
|
||||
is destroyed or reused with another document, the `std::string_view` instance becomes invalid.
|
||||
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
auto doc = parser.iterate(json);
|
||||
for(auto field : doc.get_object()) {
|
||||
std::string_view keyv = field.unescaped_key();
|
||||
@@ -670,9 +670,11 @@ in production systems:
|
||||
Some care is needed when using the On-Demand API in scenarios where you need to access several sibling arrays or objects because
|
||||
only one object or array can be active at any one time. Let us consider the following example:
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
ondemand::parser parser;
|
||||
// R"( ... )" is a C++ raw string literal.
|
||||
const padded_string json = R"({ "parent": {"child1": {"name": "John"} , "child2": {"name": "Daniel"}} })"_padded;
|
||||
// _padded returns an simdjson padded_string instance
|
||||
auto doc = parser.iterate(json);
|
||||
ondemand::object parent = doc["parent"];
|
||||
// parent owns the focus
|
||||
@@ -688,7 +690,7 @@ in production systems:
|
||||
|
||||
A correct usage is given by the following example:
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
ondemand::parser parser;
|
||||
const padded_string json = R"({ "parent": {"child1": {"name": "John"} , "child2": {"name": "Daniel"}} })"_padded;
|
||||
auto doc = parser.iterate(json);
|
||||
@@ -754,7 +756,7 @@ Some users wish to run at the best possible speed. Under recent Intel and AMD pr
|
||||
|
||||
Given that the On-Demand API offer limited runtime dispatching, it matters that your code is compiled against a specific CPU target. You should verify that the code is compiled against the target you expect. Thankfully, the simdjson library will tell you exactly what it detects as an implementation: `icelake` (AVX512 x64 processors), `haswell` (AVX2 x64 processors), `westmere` (SSE4 x64 processors), `arm64` (64-bit ARM), `ppc64` (64-bit POWER), `lasx` (LoongArch), `lsx` (LoongArch), `fallback` (others). Under x64 processors, many programmers will want to target `haswell` whereas under ARM, most programmers will want to target `arm64` (and it should do so automatically). The `fallback` is probably only good for testing purposes, not for deployment.
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
std::cout << simdjson::builtin_implementation()->name() << std::endl;
|
||||
```
|
||||
|
||||
|
||||
@@ -132,7 +132,7 @@ Whitespace Characters:
|
||||
Some official formats **(non-exhaustive list)**:
|
||||
- [Newline-Delimited JSON (NDJSON)](https://github.com/ndjson/ndjson-spec)
|
||||
- [JSON lines (JSONL)](http://jsonlines.org/)
|
||||
- [Record separator-delimited JSON (RFC 7464)](https://tools.ietf.org/html/rfc7464) <- Not supported by JsonStream!
|
||||
- [Record separator-delimited JSON (RFC 7464)](https://tools.ietf.org/html/rfc7464) <- Not supported by simdjson!
|
||||
- [More on Wikipedia...](https://en.wikipedia.org/wiki/JSON_streaming)
|
||||
|
||||
API
|
||||
@@ -184,7 +184,7 @@ You may also call the `source()` method to get a `std::string_view` instance on
|
||||
Let us illustrate the idea with code:
|
||||
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
auto json = R"([1,2,3] {"1":1,"2":3,"4":4} [1,2,3] )"_padded;
|
||||
simdjson::dom::parser parser;
|
||||
simdjson::dom::document_stream stream;
|
||||
@@ -225,7 +225,7 @@ Some users may need to work with truncated streams. The simdjson may truncate do
|
||||
|
||||
Consider the following example where a truncated document (`{"key":"intentionally unclosed string `) containing 39 bytes has been left within the stream. In such cases, the first two whole documents are parsed and returned, and the `truncated_bytes()` method returns 39.
|
||||
|
||||
```C++
|
||||
```cpp
|
||||
auto json = R"([1,2,3] {"1":1,"2":3,"4":4} {"key":"intentionally unclosed string )"_padded;
|
||||
simdjson::dom::parser parser;
|
||||
simdjson::dom::document_stream stream;
|
||||
|
||||
@@ -47,7 +47,7 @@ and reuse it. The simdjson library will allocate and retain internal buffers bet
|
||||
buffers hot in cache and keeping memory allocation and initialization to a minimum. In this manner,
|
||||
you can parse terabytes of JSON data without doing any new allocation.
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
ondemand::parser parser;
|
||||
|
||||
// This initializes buffers big enough to handle this JSON.
|
||||
@@ -71,14 +71,14 @@ Reusing string buffers
|
||||
|
||||
We recommend against creating many `std::string` or `simdjson::padded_string` instances to store the JSON content in your application. [Creating many non-trivial objects is convenient but often surprisingly slow](https://lemire.me/blog/2020/08/08/performance-tip-constructing-many-non-trivial-objects-is-slow/). Instead, as much as possible, you should allocate (once or a few times) reusable memory buffers where you write your JSON content. If you have a buffer `json_str` (of type `char*`) allocated for `capacity` bytes and you store a JSON document spanning `length` bytes, you can pass it to simdjson as follows:
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
auto doc = parser.iterate(padded_string_view(json_str, length, capacity));
|
||||
```
|
||||
|
||||
or simply
|
||||
|
||||
|
||||
```c++
|
||||
```cpp
|
||||
auto doc = parser.iterate(json_str, length, capacity);
|
||||
```
|
||||
|
||||
@@ -89,7 +89,7 @@ Server Loops: Long-Running Processes and Memory Capacity
|
||||
The On-Demand approach also automatically expands its memory capacity when larger documents are parsed. However, for longer processes where very large files are processed (such as server loops), this capacity is not resized down. On-Demand also lets you adjust the maximal capacity that the parser can process:
|
||||
|
||||
* You can set an upper bound (*max_capacity*) when construction the parser:
|
||||
```C++
|
||||
```cpp
|
||||
ondemand::parser parser(1000*1000); // Never grows past documents > 1 MB
|
||||
auto doc = parser.iterate(json);
|
||||
for (web_request request : listen()) {
|
||||
@@ -105,7 +105,7 @@ The On-Demand approach also automatically expands its memory capacity when large
|
||||
The capacity will grow as the parser encounters larger documents up to 1 MB.
|
||||
|
||||
* You can also allocate a *fixed capacity* that will never grow:
|
||||
```C++
|
||||
```cpp
|
||||
ondemand::parser parser(1000*1000);
|
||||
parser.allocate(1000*1000) // Fix the capacity to 1 MB
|
||||
auto doc = parser.iterate(json);
|
||||
@@ -297,4 +297,62 @@ int main() {
|
||||
}
|
||||
return EXIT_SUCCESS;
|
||||
}
|
||||
```
|
||||
|
||||
Further, whenever you allocate N bytes, memory allocators tend to allocate more memory, without you necessarily knowing about it. Under linux, you can use the `malloc_usable_size` function to see how much memory was actually allocated.
|
||||
Under an Apple plateform, you can `malloc_size`. The following program illustrates the usage.
|
||||
|
||||
```cpp
|
||||
#include <iostream>
|
||||
#include <cstddef>
|
||||
#include <memory>
|
||||
#include <cstdlib>
|
||||
#ifdef __APPLE__
|
||||
#include <malloc/malloc.h> // for malloc_size on macOS
|
||||
#endif
|
||||
#ifdef __linux__
|
||||
#include <malloc.h> // for malloc_usable_size on Linux
|
||||
#endif
|
||||
size_t get_usable_size(void* ptr) {
|
||||
#ifdef __linux__
|
||||
return malloc_usable_size(ptr);
|
||||
#elif defined(__APPLE__)
|
||||
return malloc_size(ptr);
|
||||
#else
|
||||
return 0; // Unsupported platform
|
||||
#endif
|
||||
}
|
||||
|
||||
int main() {
|
||||
std::cout << "Demonstrating allocation overhead and rounding with operator new\n\n";
|
||||
|
||||
#ifdef __linux__
|
||||
std::cout << "Platform: Linux\n";
|
||||
#elif defined(__APPLE__)
|
||||
std::cout << "Platform: macOS (using malloc_size)\n";
|
||||
#else
|
||||
std::cout << "Platform: Other/unsupported (usable size will show 0)\n";
|
||||
#endif
|
||||
|
||||
std::cout << "Requested size | Actual usable size\n";
|
||||
std::cout << "---------------|-------------------\n";
|
||||
size_t total_requested = 0;
|
||||
size_t total_usable = 0;
|
||||
for (size_t requested = 1; requested <= 4096; requested++) {
|
||||
total_requested += requested;
|
||||
std::unique_ptr<char[]> ptr(new char[requested]); // Allocate
|
||||
size_t usable = get_usable_size(ptr.get()); // Get usable size
|
||||
total_usable += usable;
|
||||
|
||||
std::cout << requested << "\t | " << usable << "\n";
|
||||
}
|
||||
std::cout << "---------------|-------------------\n";
|
||||
std::cout << "Total requested: " << total_requested << " bytes\n";
|
||||
std::cout << "Total usable: " << total_usable << " bytes\n";
|
||||
std::cout << "Total overhead: " << (total_usable - total_requested) << " bytes\n";
|
||||
std::cout << "Percentage overhead: "
|
||||
<< ((total_usable - total_requested) * 100.0 / total_requested) << " %\n";
|
||||
|
||||
return EXIT_SUCCESS;
|
||||
}
|
||||
```
|
||||
@@ -8,16 +8,11 @@
|
||||
* Minifies by first parsing, then minifying.
|
||||
*/
|
||||
extern "C" int LLVMFuzzerTestOneInput(const uint8_t *Data, size_t Size) {
|
||||
|
||||
auto begin = as_chars(Data);
|
||||
auto end = begin + Size;
|
||||
|
||||
std::string str(begin, end);
|
||||
simdjson::padded_string str(reinterpret_cast<const char *>(Data), Size);
|
||||
simdjson::dom::parser parser;
|
||||
simdjson::dom::element elem;
|
||||
auto error = parser.parse(str).get(elem);
|
||||
if (error) { return 0; }
|
||||
|
||||
std::string minified = simdjson::minify(elem);
|
||||
(void)minified;
|
||||
return 0;
|
||||
|
||||
@@ -35,7 +35,7 @@ cmake .. \
|
||||
-DSIMDJSON_DISABLE_DEPRECATED_API=On \
|
||||
-DSIMDJSON_FUZZ_LDFLAGS=$LIB_FUZZING_ENGINE
|
||||
|
||||
cmake --build . --target all_fuzzers
|
||||
cmake --build . --target all_fuzzers all_tests
|
||||
|
||||
cp fuzz/fuzz_* $OUT
|
||||
|
||||
|
||||
|
After Width: | Height: | Size: 39 KiB |
|
After Width: | Height: | Size: 4.2 KiB |
|
After Width: | Height: | Size: 35 KiB |
|
After Width: | Height: | Size: 93 KiB |
|
After Width: | Height: | Size: 226 KiB |
|
After Width: | Height: | Size: 136 KiB |
|
After Width: | Height: | Size: 46 KiB |
|
After Width: | Height: | Size: 108 KiB |
|
After Width: | Height: | Size: 256 KiB |
|
After Width: | Height: | Size: 136 KiB |
|
After Width: | Height: | Size: 46 KiB |
|
After Width: | Height: | Size: 109 KiB |
|
After Width: | Height: | Size: 258 KiB |
|
After Width: | Height: | Size: 136 KiB |
@@ -52,7 +52,13 @@
|
||||
#include "simdjson/padded_string_view-inl.h"
|
||||
|
||||
#include "simdjson/dom.h"
|
||||
#include "simdjson/builder.h"
|
||||
#include "simdjson/ondemand.h"
|
||||
#include "simdjson/convert.h"
|
||||
#include "simdjson/convert-inl.h"
|
||||
|
||||
// Compile-time JSON parsing (C++26 P2996 reflection)
|
||||
#include "simdjson/compile_time_json.h"
|
||||
#include "simdjson/compile_time_json-inl.h"
|
||||
|
||||
#endif // SIMDJSON_H
|
||||
|
||||
@@ -0,0 +1,8 @@
|
||||
#ifndef SIMDJSON_ARM64_BUILDER_H
|
||||
#define SIMDJSON_ARM64_BUILDER_H
|
||||
|
||||
#include "simdjson/arm64/begin.h"
|
||||
#include "simdjson/generic/builder/amalgamated.h"
|
||||
#include "simdjson/arm64/end.h"
|
||||
|
||||
#endif // SIMDJSON_ARM64_BUILDER_H
|
||||
@@ -17,6 +17,7 @@ using namespace simd;
|
||||
struct backslash_and_quote {
|
||||
public:
|
||||
static constexpr uint32_t BYTES_PROCESSED = 32;
|
||||
// We only copy if dst is non-null.
|
||||
simdjson_inline backslash_and_quote copy_and_find(const uint8_t *src, uint8_t *dst);
|
||||
|
||||
simdjson_inline bool has_quote_first() { return ((bs_bits - 1) & quote_bits) != 0; }
|
||||
@@ -34,8 +35,10 @@ simdjson_inline backslash_and_quote backslash_and_quote::copy_and_find(const uin
|
||||
static_assert(SIMDJSON_PADDING >= (BYTES_PROCESSED - 1), "backslash and quote finder must process fewer than SIMDJSON_PADDING bytes");
|
||||
simd8<uint8_t> v0(src);
|
||||
simd8<uint8_t> v1(src + sizeof(v0));
|
||||
v0.store(dst);
|
||||
v1.store(dst + sizeof(v0));
|
||||
if(dst != nullptr) {
|
||||
v0.store(dst);
|
||||
v1.store(dst + sizeof(v0));
|
||||
}
|
||||
|
||||
// Getting a 64-bit bitmask is much cheaper than multiple 16-bit bitmasks on ARM; therefore, we
|
||||
// smash them together into a 64-byte mask and get the bitmask from there.
|
||||
|
||||
@@ -0,0 +1,14 @@
|
||||
#ifndef SIMDJSON_BUILDER_H
|
||||
#define SIMDJSON_BUILDER_H
|
||||
|
||||
#include "simdjson/builtin/builder.h"
|
||||
|
||||
namespace simdjson {
|
||||
/**
|
||||
* @copydoc simdjson::builtin::builder
|
||||
*/
|
||||
namespace builder = builtin::builder;
|
||||
|
||||
} // namespace simdjson
|
||||
|
||||
#endif // SIMDJSON_BUILDER_H
|
||||
@@ -20,10 +20,10 @@
|
||||
#include "simdjson/ppc64.h"
|
||||
#elif SIMDJSON_BUILTIN_IMPLEMENTATION_IS(westmere)
|
||||
#include "simdjson/westmere.h"
|
||||
#elif SIMDJSON_BUILTIN_IMPLEMENTATION_IS(lsx)
|
||||
#include "simdjson/lsx.h"
|
||||
#elif SIMDJSON_BUILTIN_IMPLEMENTATION_IS(lasx)
|
||||
#include "simdjson/lasx.h"
|
||||
#elif SIMDJSON_BUILTIN_IMPLEMENTATION_IS(lsx)
|
||||
#include "simdjson/lsx.h"
|
||||
#else
|
||||
#error Unknown SIMDJSON_BUILTIN_IMPLEMENTATION
|
||||
#endif
|
||||
|
||||
@@ -0,0 +1,40 @@
|
||||
#ifndef SIMDJSON_BUILTIN_BUILDER_H
|
||||
#define SIMDJSON_BUILTIN_BUILDER_H
|
||||
|
||||
#include "simdjson/builtin.h"
|
||||
#include "simdjson/builtin/base.h"
|
||||
|
||||
#include "simdjson/generic/builder/dependencies.h"
|
||||
|
||||
#define SIMDJSON_CONDITIONAL_INCLUDE
|
||||
|
||||
#if SIMDJSON_BUILTIN_IMPLEMENTATION_IS(arm64)
|
||||
#include "simdjson/arm64/builder.h"
|
||||
#elif SIMDJSON_BUILTIN_IMPLEMENTATION_IS(fallback)
|
||||
#include "simdjson/fallback/builder.h"
|
||||
#elif SIMDJSON_BUILTIN_IMPLEMENTATION_IS(haswell)
|
||||
#include "simdjson/haswell/builder.h"
|
||||
#elif SIMDJSON_BUILTIN_IMPLEMENTATION_IS(icelake)
|
||||
#include "simdjson/icelake/builder.h"
|
||||
#elif SIMDJSON_BUILTIN_IMPLEMENTATION_IS(ppc64)
|
||||
#include "simdjson/ppc64/builder.h"
|
||||
#elif SIMDJSON_BUILTIN_IMPLEMENTATION_IS(westmere)
|
||||
#include "simdjson/westmere/builder.h"
|
||||
#elif SIMDJSON_BUILTIN_IMPLEMENTATION_IS(lsx)
|
||||
#include "simdjson/lsx/builder.h"
|
||||
#elif SIMDJSON_BUILTIN_IMPLEMENTATION_IS(lasx)
|
||||
#include "simdjson/lasx/builder.h"
|
||||
#else
|
||||
#error Unknown SIMDJSON_BUILTIN_IMPLEMENTATION
|
||||
#endif
|
||||
|
||||
#undef SIMDJSON_CONDITIONAL_INCLUDE
|
||||
|
||||
namespace simdjson {
|
||||
/**
|
||||
* @copydoc simdjson::SIMDJSON_BUILTIN_IMPLEMENTATION::builder
|
||||
*/
|
||||
namespace builder = SIMDJSON_BUILTIN_IMPLEMENTATION::builder;
|
||||
} // namespace simdjson
|
||||
|
||||
#endif // SIMDJSON_BUILTIN_BUILDER_H
|
||||
@@ -289,7 +289,9 @@ namespace std {
|
||||
// when the compiler is optimizing.
|
||||
// We only set SIMDJSON_DEVELOPMENT_CHECKS if both __OPTIMIZE__
|
||||
// and NDEBUG are not defined.
|
||||
#if !defined(__OPTIMIZE__) && !defined(NDEBUG)
|
||||
// We recognize _DEBUG as overriding __OPTIMIZE__ so that if both
|
||||
// __OPTIMIZE__ and _DEBUG are defined, we still set SIMDJSON_DEVELOPMENT_CHECKS.
|
||||
#if ((!defined(__OPTIMIZE__) || defined(_DEBUG)) && !defined(NDEBUG))
|
||||
#define SIMDJSON_DEVELOPMENT_CHECKS 1
|
||||
#endif // __OPTIMIZE__
|
||||
#endif // _MSC_VER
|
||||
|
||||
@@ -0,0 +1,75 @@
|
||||
/**
|
||||
* @file compile_time_json.h
|
||||
* @brief Compile-time JSON parsing using C++26 reflection with
|
||||
* std::meta::substitute()
|
||||
*/
|
||||
|
||||
#ifndef SIMDJSON_GENERIC_COMPILE_TIME_JSON_H
|
||||
#define SIMDJSON_GENERIC_COMPILE_TIME_JSON_H
|
||||
|
||||
#if SIMDJSON_STATIC_REFLECTION
|
||||
|
||||
#include <algorithm>
|
||||
#include <array>
|
||||
#include <charconv>
|
||||
#include <cstdint>
|
||||
#include <expected>
|
||||
#include <meta>
|
||||
#include <string>
|
||||
#include <string_view>
|
||||
#include <vector>
|
||||
|
||||
namespace simdjson {
|
||||
namespace compile_time {
|
||||
|
||||
/**
|
||||
* @brief Compile-time JSON parser. This function parses the provided JSON
|
||||
* string at compile time and returns a custom struct type representing the JSON
|
||||
* object.
|
||||
*
|
||||
* We have a few limitations which trigger compile-time errors if violated:
|
||||
* - Only JSON objects and arrays are supported at the top level (no primitives).
|
||||
* We will lift this limitation in the future.
|
||||
* - Strings are represented using the const char * in UTF-8, but they must not
|
||||
* contain embedded nulls. We would prefer to represent them as std::string or
|
||||
* std::string_view, and hope to do so in the future.
|
||||
* - Heterogeneous arrays are not supported yet. E.g., you need to have arrays of
|
||||
* all integers, or all strings, all floats, all compatible objects, etc.
|
||||
* For example, the following is accepted:
|
||||
* [
|
||||
* { "name": "Alice", "age": 30 },
|
||||
* { "name": "Bob", "age": 25 },
|
||||
* { "name": "Charlie", "age": 35 }
|
||||
* ]
|
||||
* but the following is not:
|
||||
* [
|
||||
* { "name": "Alice", "age": 30 },
|
||||
* "Just a string",
|
||||
* 42,
|
||||
* { "name": "Charlie", "age": 35 }
|
||||
* ]
|
||||
*
|
||||
* We may support heterogeneous arrays in the future with std::variant types.
|
||||
* - We parse the first JSON document encountered in the string. Trailing
|
||||
* characters are ignored. Thus if your JSON begins with {"a":1}, everything
|
||||
* after the closing } is ignored. This limitation will be lifted in the future,
|
||||
* reporting an error.
|
||||
*
|
||||
* These limitations are safe in the sense that they result in compile-time errors.
|
||||
* Thus you will not get truncated strings or imprecise floats silently.
|
||||
*
|
||||
* This function is subject to change in the future.
|
||||
*/
|
||||
template <constevalutil::fixed_string json_str> consteval auto parse_json();
|
||||
|
||||
} // namespace compile_time
|
||||
} // namespace simdjson
|
||||
|
||||
|
||||
template <simdjson::constevalutil::fixed_string str>
|
||||
consteval auto operator ""_json() {
|
||||
return simdjson::compile_time::parse_json<str>();
|
||||
}
|
||||
|
||||
#endif // SIMDJSON_STATIC_REFLECTION
|
||||
#endif // SIMDJSON_GENERIC_COMPILE_TIME_JSON_H
|
||||
@@ -13,6 +13,11 @@
|
||||
#endif
|
||||
#endif
|
||||
|
||||
// C++ 26
|
||||
#if !defined(SIMDJSON_CPLUSPLUS26) && (SIMDJSON_CPLUSPLUS >= 202402L) // update when the standard is finalized
|
||||
#define SIMDJSON_CPLUSPLUS26 1
|
||||
#endif
|
||||
|
||||
// C++ 23
|
||||
#if !defined(SIMDJSON_CPLUSPLUS23) && (SIMDJSON_CPLUSPLUS >= 202302L)
|
||||
#define SIMDJSON_CPLUSPLUS23 1
|
||||
|
||||
@@ -143,6 +143,25 @@ concept container_but_not_string =
|
||||
std::ranges::input_range<T> && !string_like<T> && !concepts::string_view_keyed_map<T>;
|
||||
|
||||
|
||||
|
||||
// Concept: Indexable container that is not a string or associative container
|
||||
// Accepts: std::vector, std::array, std::deque (have operator[], value_type, not string_like)
|
||||
// Rejects: std::string (string_like), std::list (no operator[]), std::map (has key_type)
|
||||
template<typename Container>
|
||||
concept indexable_container = requires {
|
||||
typename Container::value_type;
|
||||
requires !concepts::string_like<Container>;
|
||||
requires !requires { typename Container::key_type; }; // Reject maps/sets
|
||||
requires requires(Container& c, std::size_t i) {
|
||||
{ c[i] } -> std::convertible_to<typename Container::value_type>;
|
||||
};
|
||||
};
|
||||
|
||||
|
||||
// Variable template to use with std::meta::substitute
|
||||
template<typename Container>
|
||||
constexpr bool indexable_container_v = indexable_container<Container>;
|
||||
|
||||
} // namespace concepts
|
||||
|
||||
|
||||
|
||||
@@ -51,7 +51,7 @@ consteval std::string consteval_to_quoted_escaped(std::string_view input) {
|
||||
|
||||
|
||||
#if SIMDJSON_SUPPORTS_CONCEPTS
|
||||
template <std::size_t N>
|
||||
template <size_t N>
|
||||
struct fixed_string {
|
||||
constexpr fixed_string(const char (&str)[N]) {
|
||||
for (std::size_t i = 0; i < N; ++i) {
|
||||
@@ -60,6 +60,20 @@ struct fixed_string {
|
||||
}
|
||||
char data[N];
|
||||
constexpr std::string_view view() const { return {data, N - 1}; }
|
||||
constexpr size_t size() const { return N ; }
|
||||
constexpr operator std::string_view() const { return view(); }
|
||||
constexpr char operator[](std::size_t index) const { return data[index]; }
|
||||
constexpr bool operator==(const fixed_string& other) const {
|
||||
if (N != other.size()) {
|
||||
return false;
|
||||
}
|
||||
for (std::size_t i = 0; i < N; ++i) {
|
||||
if (data[i] != other.data[i]) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
};
|
||||
template <std::size_t N>
|
||||
fixed_string(const char (&)[N]) -> fixed_string<N>;
|
||||
|
||||
@@ -9,6 +9,7 @@
|
||||
#include "simdjson/dom/object.h"
|
||||
#include "simdjson/dom/parser.h"
|
||||
#include "simdjson/dom/serialization.h"
|
||||
#include "simdjson/dom/fractured_json.h"
|
||||
|
||||
// Inline functions
|
||||
#include "simdjson/dom/array-inl.h"
|
||||
@@ -19,5 +20,6 @@
|
||||
#include "simdjson/dom/parser-inl.h"
|
||||
#include "simdjson/internal/tape_ref-inl.h"
|
||||
#include "simdjson/dom/serialization-inl.h"
|
||||
#include "simdjson/dom/fractured_json-inl.h"
|
||||
|
||||
#endif // SIMDJSON_DOM_H
|
||||
|
||||
@@ -111,7 +111,7 @@ public:
|
||||
inline simdjson_result<element> at_pointer(std::string_view json_pointer) const noexcept;
|
||||
|
||||
/**
|
||||
* Recursive function which processes the json path of each child element
|
||||
* Recursive function which processes the JSON path of each child element
|
||||
*/
|
||||
inline void process_json_path_of_child_elements(std::vector<element>::iterator& current, std::vector<element>::iterator& end, const std::string_view& path_suffix, std::vector<element>& accumulator) const noexcept;
|
||||
|
||||
@@ -126,7 +126,7 @@ public:
|
||||
* JSONPath queries that trivially convertible to JSON Pointer queries: key
|
||||
* names and array indices.
|
||||
*
|
||||
* https://datatracker.ietf.org/doc/html/draft-normington-jsonpath-00
|
||||
* https://www.rfc-editor.org/rfc/rfc9535 (RFC 9535)
|
||||
*
|
||||
* @return The value associated with the given JSONPath expression, or:
|
||||
* - INVALID_JSON_POINTER if the JSONPath to JSON Pointer conversion fails
|
||||
|
||||
@@ -74,7 +74,7 @@ public:
|
||||
/**
|
||||
* Construct an uninitialized document_stream.
|
||||
*
|
||||
* ```c++
|
||||
* ```cpp
|
||||
* document_stream docs;
|
||||
* error = parser.parse_many(json).get(docs);
|
||||
* ```
|
||||
|
||||
@@ -408,7 +408,7 @@ public:
|
||||
* JSONPath queries that trivially convertible to JSON Pointer queries: key
|
||||
* names and array indices.
|
||||
*
|
||||
* https://datatracker.ietf.org/doc/html/draft-normington-jsonpath-00
|
||||
* https://www.rfc-editor.org/rfc/rfc9535 (RFC 9535)
|
||||
*
|
||||
* @return The value associated with the given JSONPath expression, or:
|
||||
* - INVALID_JSON_POINTER if the JSONPath to JSON Pointer conversion fails
|
||||
|
||||
@@ -0,0 +1,159 @@
|
||||
#ifndef SIMDJSON_DOM_FRACTURED_JSON_H
|
||||
#define SIMDJSON_DOM_FRACTURED_JSON_H
|
||||
|
||||
#include "simdjson/dom/base.h"
|
||||
#include "simdjson/dom/element.h"
|
||||
|
||||
namespace simdjson {
|
||||
|
||||
/**
|
||||
* Configuration options for FracturedJson formatting.
|
||||
*
|
||||
* FracturedJson intelligently chooses between different layout strategies
|
||||
* (inline, compact multiline, table, expanded) based on content complexity,
|
||||
* length, and structure similarity.
|
||||
*/
|
||||
struct fractured_json_options {
|
||||
/**
|
||||
* Maximum total characters per line (default: 120).
|
||||
* Content exceeding this will be expanded to multiple lines.
|
||||
*/
|
||||
size_t max_total_line_length = 120;
|
||||
|
||||
/**
|
||||
* Maximum length for inlined elements (default: 80).
|
||||
* Simple arrays/objects shorter than this may be rendered inline.
|
||||
*/
|
||||
size_t max_inline_length = 80;
|
||||
|
||||
/**
|
||||
* Maximum nesting depth for inline rendering (default: 2).
|
||||
* Elements with complexity exceeding this will be expanded.
|
||||
* Complexity 0 = scalar, 1 = flat array/object, 2 = one level of nesting.
|
||||
*/
|
||||
size_t max_inline_complexity = 2;
|
||||
|
||||
/**
|
||||
* Maximum complexity for compact array formatting (default: 1).
|
||||
* Arrays with elements of this complexity or less may have multiple
|
||||
* items per line.
|
||||
*/
|
||||
size_t max_compact_array_complexity = 1;
|
||||
|
||||
/**
|
||||
* Number of spaces per indentation level (default: 4).
|
||||
*/
|
||||
size_t indent_spaces = 4;
|
||||
|
||||
/**
|
||||
* Enable tabular formatting for arrays of similar objects (default: true).
|
||||
* When enabled, arrays of objects with identical keys are formatted
|
||||
* as aligned tables.
|
||||
*/
|
||||
bool enable_table_format = true;
|
||||
|
||||
/**
|
||||
* Minimum number of rows to trigger table mode (default: 3).
|
||||
*/
|
||||
size_t min_table_rows = 3;
|
||||
|
||||
/**
|
||||
* Similarity threshold for table detection (default: 0.8).
|
||||
* Objects must share at least this fraction of keys to be formatted
|
||||
* as a table.
|
||||
*/
|
||||
double table_similarity_threshold = 0.8;
|
||||
|
||||
/**
|
||||
* Enable compact multiline arrays (default: true).
|
||||
* When enabled, arrays of simple elements may have multiple items
|
||||
* per line.
|
||||
*/
|
||||
bool enable_compact_multiline = true;
|
||||
|
||||
/**
|
||||
* Maximum array items per line in compact mode (default: 10).
|
||||
*/
|
||||
size_t max_items_per_line = 10;
|
||||
|
||||
/**
|
||||
* Add space inside brackets for simple containers (default: true).
|
||||
* When true: { "key": "value" }
|
||||
* When false: {"key": "value"}
|
||||
*/
|
||||
bool simple_bracket_padding = true;
|
||||
|
||||
/**
|
||||
* Add space after colons (default: true).
|
||||
* When true: "key": "value"
|
||||
* When false: "key":"value"
|
||||
*/
|
||||
bool colon_padding = true;
|
||||
|
||||
/**
|
||||
* Add space after commas in inline content (default: true).
|
||||
* When true: [1, 2, 3]
|
||||
* When false: [1,2,3]
|
||||
*/
|
||||
bool comma_padding = true;
|
||||
};
|
||||
|
||||
/**
|
||||
* Format JSON using FracturedJson formatting with default options.
|
||||
*
|
||||
* FracturedJson produces human-readable yet compact output by intelligently
|
||||
* choosing between inline, compact multiline, table, and expanded layouts.
|
||||
*
|
||||
* dom::parser parser;
|
||||
* element doc = parser.parse(json_string);
|
||||
* cout << fractured_json(doc) << endl;
|
||||
*/
|
||||
template <class T>
|
||||
std::string fractured_json(T x);
|
||||
|
||||
/**
|
||||
* Format JSON using FracturedJson formatting with custom options.
|
||||
*
|
||||
* dom::parser parser;
|
||||
* element doc = parser.parse(json_string);
|
||||
* fractured_json_options opts;
|
||||
* opts.max_total_line_length = 80;
|
||||
* cout << fractured_json(doc, opts) << endl;
|
||||
*/
|
||||
template <class T>
|
||||
std::string fractured_json(T x, const fractured_json_options& options);
|
||||
|
||||
#if SIMDJSON_EXCEPTIONS
|
||||
template <class T>
|
||||
std::string fractured_json(simdjson_result<T> x);
|
||||
|
||||
template <class T>
|
||||
std::string fractured_json(simdjson_result<T> x, const fractured_json_options& options);
|
||||
#endif
|
||||
|
||||
/**
|
||||
* Format a JSON string using FracturedJson formatting.
|
||||
*
|
||||
* This is useful for formatting output from the builder/static reflection API
|
||||
* or any valid JSON string.
|
||||
*
|
||||
* // With static reflection
|
||||
* MyStruct data = {...};
|
||||
* auto minified = simdjson::to_json_string(data);
|
||||
* auto formatted = simdjson::fractured_json_string(minified.value());
|
||||
*
|
||||
* // Or with any JSON string
|
||||
* std::string json = R"({"key":"value"})";
|
||||
* auto formatted = simdjson::fractured_json_string(json);
|
||||
*/
|
||||
inline std::string fractured_json_string(std::string_view json_str);
|
||||
|
||||
/**
|
||||
* Format a JSON string using FracturedJson formatting with custom options.
|
||||
*/
|
||||
inline std::string fractured_json_string(std::string_view json_str,
|
||||
const fractured_json_options& options);
|
||||
|
||||
} // namespace simdjson
|
||||
|
||||
#endif // SIMDJSON_DOM_FRACTURED_JSON_H
|
||||
@@ -186,7 +186,7 @@ inline simdjson_result<std::vector<element>> object::at_path_with_wildcard(std::
|
||||
}
|
||||
|
||||
if (i >= json_path.size() || (json_path[i] != '.' && json_path[i] != '[')) {
|
||||
// expect json path to always start with $ but this isn't currently
|
||||
// expect JSONPath expressions to always start with $ but this isn't currently
|
||||
// expected in jsonpathutil.h.
|
||||
return INVALID_JSON_POINTER;
|
||||
}
|
||||
|
||||
@@ -175,7 +175,7 @@ public:
|
||||
inline simdjson_result<element> at_pointer(std::string_view json_pointer) const noexcept;
|
||||
|
||||
/**
|
||||
* Recursive function which processes the json path of each child element
|
||||
* Recursive function which processes the JSON path of each child element
|
||||
*/
|
||||
inline void process_json_path_of_child_elements(std::vector<element>::iterator& current, std::vector<element>::iterator& end, const std::string_view& path_suffix, std::vector<element>& accumulator) const noexcept;
|
||||
|
||||
@@ -189,7 +189,7 @@ public:
|
||||
* JSONPath queries that trivially convertible to JSON Pointer queries: key
|
||||
* names and array indices.
|
||||
*
|
||||
* https://datatracker.ietf.org/doc/html/draft-normington-jsonpath-00
|
||||
* https://www.rfc-editor.org/rfc/rfc9535 (RFC 9535)
|
||||
*
|
||||
* @return The value associated with the given JSONPath expression, or:
|
||||
* - INVALID_JSON_POINTER if the JSONPath to JSON Pointer conversion fails
|
||||
|
||||
@@ -204,10 +204,7 @@ public:
|
||||
*
|
||||
* ### std::string references
|
||||
*
|
||||
* If you pass a mutable std::string reference (std::string&), the parser will seek to extend
|
||||
* its capacity to SIMDJSON_PADDING bytes beyond the end of the string.
|
||||
*
|
||||
* Whenever you pass an std::string reference, the parser will access the bytes beyond the end of
|
||||
* Whenever you pass an std::string reference, the parser may access the bytes beyond the end of
|
||||
* the string but before the end of the allocated memory (std::string::capacity()).
|
||||
* If you are using a sanitizer that checks for reading uninitialized bytes or std::string's
|
||||
* container-overflow checks, you may encounter sanitizer warnings.
|
||||
@@ -239,7 +236,7 @@ public:
|
||||
/** @overload parse(const uint8_t *buf, size_t len, bool realloc_if_needed) */
|
||||
simdjson_inline simdjson_result<element> parse(const char *buf, size_t len, bool realloc_if_needed = true) & noexcept;
|
||||
simdjson_inline simdjson_result<element> parse(const char *buf, size_t len, bool realloc_if_needed = true) && =delete;
|
||||
/** @overload parse(const uint8_t *buf, size_t len, bool realloc_if_needed) */
|
||||
/** @overload parse(const std::string &) */
|
||||
simdjson_inline simdjson_result<element> parse(const std::string &s) & noexcept;
|
||||
simdjson_inline simdjson_result<element> parse(const std::string &s) && =delete;
|
||||
/** @overload parse(const uint8_t *buf, size_t len, bool realloc_if_needed) */
|
||||
|
||||
@@ -8,7 +8,7 @@
|
||||
namespace simdjson {
|
||||
|
||||
inline bool is_fatal(error_code error) noexcept {
|
||||
return error == TAPE_ERROR || error == INCOMPLETE_ARRAY_OR_OBJECT;
|
||||
return error == TAPE_ERROR || error == INCOMPLETE_ARRAY_OR_OBJECT || error == OUT_OF_ORDER_ITERATION || error == DEPTH_ERROR;
|
||||
}
|
||||
|
||||
namespace internal {
|
||||
|
||||
@@ -265,6 +265,8 @@ struct simdjson_result_base : protected std::pair<T, error_code> {
|
||||
*/
|
||||
simdjson_inline T&& value_unsafe() && noexcept;
|
||||
|
||||
using value_type = T;
|
||||
using error_type = error_code;
|
||||
}; // struct simdjson_result_base
|
||||
|
||||
} // namespace internal
|
||||
@@ -376,6 +378,8 @@ struct simdjson_result : public internal::simdjson_result_base<T> {
|
||||
*/
|
||||
simdjson_inline T&& value_unsafe() && noexcept;
|
||||
|
||||
using value_type = T;
|
||||
using error_type = error_code;
|
||||
}; // struct simdjson_result
|
||||
|
||||
#if SIMDJSON_EXCEPTIONS
|
||||
|
||||
@@ -0,0 +1,8 @@
|
||||
#ifndef SIMDJSON_FALLBACK_BUILDER_H
|
||||
#define SIMDJSON_FALLBACK_BUILDER_H
|
||||
|
||||
#include "simdjson/fallback/begin.h"
|
||||
#include "simdjson/generic/builder/amalgamated.h"
|
||||
#include "simdjson/fallback/end.h"
|
||||
|
||||
#endif // SIMDJSON_FALLBACK_BUILDER_H
|
||||
@@ -13,6 +13,7 @@ namespace {
|
||||
struct backslash_and_quote {
|
||||
public:
|
||||
static constexpr uint32_t BYTES_PROCESSED = 1;
|
||||
// We only copy if dst is non-null.
|
||||
simdjson_inline backslash_and_quote copy_and_find(const uint8_t *src, uint8_t *dst);
|
||||
|
||||
simdjson_inline bool has_quote_first() { return c == '"'; }
|
||||
@@ -25,7 +26,9 @@ public:
|
||||
|
||||
simdjson_inline backslash_and_quote backslash_and_quote::copy_and_find(const uint8_t *src, uint8_t *dst) {
|
||||
// store to dest unconditionally - we can overwrite the bits we don't like later
|
||||
dst[0] = src[0];
|
||||
if(dst != nullptr) {
|
||||
dst[0] = src[0];
|
||||
}
|
||||
return { src[0] };
|
||||
}
|
||||
|
||||
|
||||
@@ -17,10 +17,10 @@
|
||||
#include "simdjson/arm64/begin.h"
|
||||
#elif SIMDJSON_IMPLEMENTATION_PPC64
|
||||
#include "simdjson/ppc64/begin.h"
|
||||
#elif SIMDJSON_IMPLEMENTATION_LSX
|
||||
#include "simdjson/lsx/begin.h"
|
||||
#elif SIMDJSON_IMPLEMENTATION_LASX
|
||||
#include "simdjson/lasx/begin.h"
|
||||
#elif SIMDJSON_IMPLEMENTATION_LSX
|
||||
#include "simdjson/lsx/begin.h"
|
||||
#elif SIMDJSON_IMPLEMENTATION_FALLBACK
|
||||
#include "simdjson/fallback/begin.h"
|
||||
#else
|
||||
|
||||
@@ -0,0 +1,13 @@
|
||||
#if defined(SIMDJSON_CONDITIONAL_INCLUDE) && !defined(SIMDJSON_GENERIC_BUILDER_DEPENDENCIES_H)
|
||||
#error simdjson/generic/builder/dependencies.h must be included before simdjson/generic/builder/amalgamated.h!
|
||||
#endif
|
||||
|
||||
#include "simdjson/generic/builder/json_string_builder.h"
|
||||
#include "simdjson/generic/builder/json_builder.h"
|
||||
#include "simdjson/generic/builder/fractured_json_builder.h"
|
||||
|
||||
|
||||
|
||||
// JSON builder inline definitions
|
||||
#include "simdjson/generic/builder/json_string_builder-inl.h"
|
||||
|
||||
@@ -0,0 +1,14 @@
|
||||
#ifdef SIMDJSON_CONDITIONAL_INCLUDE
|
||||
#error simdjson/generic/builder/dependencies.h must be included before defining SIMDJSON_CONDITIONAL_INCLUDE!
|
||||
#endif
|
||||
|
||||
#ifndef SIMDJSON_GENERIC_BUILDER_DEPENDENCIES_H
|
||||
#define SIMDJSON_GENERIC_BUILDER_DEPENDENCIES_H
|
||||
|
||||
// Internal headers needed for builder generics.
|
||||
// All includes not under simdjson/generic/builder must be here!
|
||||
// Otherwise, amalgamation will fail.
|
||||
#include "simdjson/concepts.h"
|
||||
#include "simdjson/dom/fractured_json.h"
|
||||
|
||||
#endif // SIMDJSON_GENERIC_BUILDER_DEPENDENCIES_H
|
||||
@@ -0,0 +1,117 @@
|
||||
#ifndef SIMDJSON_GENERIC_FRACTURED_JSON_BUILDER_H
|
||||
#define SIMDJSON_GENERIC_FRACTURED_JSON_BUILDER_H
|
||||
|
||||
#ifndef SIMDJSON_CONDITIONAL_INCLUDE
|
||||
#include "simdjson/generic/builder/json_builder.h"
|
||||
#include "simdjson/dom/fractured_json.h"
|
||||
#endif // SIMDJSON_CONDITIONAL_INCLUDE
|
||||
|
||||
#if SIMDJSON_STATIC_REFLECTION
|
||||
|
||||
namespace simdjson {
|
||||
namespace SIMDJSON_IMPLEMENTATION {
|
||||
namespace builder {
|
||||
|
||||
/**
|
||||
* Serialize an object to a FracturedJson-formatted string.
|
||||
*
|
||||
* FracturedJson produces human-readable yet compact JSON output by intelligently
|
||||
* choosing between different layout strategies (inline, compact multiline, table,
|
||||
* expanded) based on content complexity, length, and structure similarity.
|
||||
*
|
||||
* This function combines the builder's serialization with FracturedJson formatting:
|
||||
* 1. Serializes the object to minified JSON using reflection
|
||||
* 2. Parses and reformats using FracturedJson
|
||||
*
|
||||
* Example:
|
||||
* struct User { int id; std::string name; bool active; };
|
||||
* User user{1, "Alice", true};
|
||||
* auto result = to_fractured_json_string(user);
|
||||
* // result.value() == "{ \"id\": 1, \"name\": \"Alice\", \"active\": true }"
|
||||
*
|
||||
* @param obj The object to serialize (must be a reflectable type)
|
||||
* @param opts FracturedJson formatting options
|
||||
* @param initial_capacity Initial buffer capacity for serialization
|
||||
* @return The formatted JSON string, or an error
|
||||
*/
|
||||
template <class T>
|
||||
simdjson_warn_unused simdjson_result<std::string> to_fractured_json_string(
|
||||
const T& obj,
|
||||
const fractured_json_options& opts = {},
|
||||
size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
|
||||
// Step 1: Serialize to minified JSON
|
||||
std::string formatted;
|
||||
auto error = to_json_string(obj, initial_capacity).get(formatted);
|
||||
if (error) {
|
||||
return error;
|
||||
}
|
||||
|
||||
// Step 2: Reformat with FracturedJson
|
||||
return fractured_json_string(formatted, opts);
|
||||
}
|
||||
|
||||
/**
|
||||
* Extract specific fields from an object and format with FracturedJson.
|
||||
*
|
||||
* Example:
|
||||
* struct User { int id; std::string name; std::string email; bool active; };
|
||||
* User user{1, "Alice", "alice@example.com", true};
|
||||
* auto result = extract_fractured_json<"id", "name">(user);
|
||||
* // result.value() == "{ \"id\": 1, \"name\": \"Alice\" }"
|
||||
*
|
||||
* @param obj The object to serialize
|
||||
* @param opts FracturedJson formatting options
|
||||
* @param initial_capacity Initial buffer capacity for serialization
|
||||
* @return The formatted JSON string containing only the specified fields
|
||||
*/
|
||||
template<constevalutil::fixed_string... FieldNames, typename T>
|
||||
requires(std::is_class_v<T> && (sizeof...(FieldNames) > 0))
|
||||
simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
|
||||
const T& obj,
|
||||
const fractured_json_options& opts = {},
|
||||
size_t initial_capacity = string_builder::DEFAULT_INITIAL_CAPACITY) {
|
||||
// Step 1: Extract fields to minified JSON
|
||||
std::string formatted;
|
||||
auto error = extract_from<FieldNames...>(obj, initial_capacity).get(formatted);
|
||||
if (error) {
|
||||
return error;
|
||||
}
|
||||
|
||||
// Step 2: Reformat with FracturedJson
|
||||
return fractured_json_string(formatted, opts);
|
||||
}
|
||||
|
||||
} // namespace builder
|
||||
} // namespace SIMDJSON_IMPLEMENTATION
|
||||
|
||||
// Global namespace convenience functions
|
||||
|
||||
/**
|
||||
* Serialize an object to a FracturedJson-formatted string.
|
||||
* Global namespace version for convenience.
|
||||
*/
|
||||
template <class T>
|
||||
simdjson_warn_unused simdjson_result<std::string> to_fractured_json_string(
|
||||
const T& obj,
|
||||
const fractured_json_options& opts = {},
|
||||
size_t initial_capacity = SIMDJSON_IMPLEMENTATION::builder::string_builder::DEFAULT_INITIAL_CAPACITY) {
|
||||
return SIMDJSON_IMPLEMENTATION::builder::to_fractured_json_string(obj, opts, initial_capacity);
|
||||
}
|
||||
/**
|
||||
* Extract specific fields from an object and format with FracturedJson.
|
||||
* Global namespace version for convenience.
|
||||
*/
|
||||
template<constevalutil::fixed_string... FieldNames, typename T>
|
||||
requires(std::is_class_v<T> && (sizeof...(FieldNames) > 0))
|
||||
simdjson_warn_unused simdjson_result<std::string> extract_fractured_json(
|
||||
const T& obj,
|
||||
const fractured_json_options& opts = {},
|
||||
size_t initial_capacity = SIMDJSON_IMPLEMENTATION::builder::string_builder::DEFAULT_INITIAL_CAPACITY) {
|
||||
return SIMDJSON_IMPLEMENTATION::builder::extract_fractured_json<FieldNames...>(obj, opts, initial_capacity);
|
||||
}
|
||||
|
||||
} // namespace simdjson
|
||||
|
||||
#endif // SIMDJSON_STATIC_REFLECTION
|
||||
|
||||
#endif // SIMDJSON_GENERIC_FRACTURED_JSON_BUILDER_H
|
||||
@@ -1,7 +1,3 @@
|
||||
/**
|
||||
* This file is part of the builder API. It is temporarily in the ondemand directory
|
||||
* but we will move it to a builder directory later.
|
||||
*/
|
||||
#ifndef SIMDJSON_GENERIC_BUILDER_H
|
||||
|
||||
#ifndef SIMDJSON_CONDITIONAL_INCLUDE
|
||||
@@ -1,7 +1,3 @@
|
||||
/**
|
||||
* This file is part of the builder API. It is temporarily in the ondemand
|
||||
* directory but we will move it to a builder directory later.
|
||||
*/
|
||||
#include <array>
|
||||
#include <cstring>
|
||||
#include <type_traits>
|
||||
@@ -416,9 +412,9 @@ simdjson_inline void string_builder::append(number_type v) noexcept {
|
||||
pv = 0 - pv; // the 0 is for Microsoft
|
||||
}
|
||||
size_t dc = internal::digit_count(pv);
|
||||
if (negative) {
|
||||
buffer.get()[position++] = '-';
|
||||
}
|
||||
// by always writing the minus sign, we avoid the branch.
|
||||
buffer.get()[position] = '-';
|
||||
position += negative ? 1 : 0;
|
||||
char *write_pointer = buffer.get() + position + dc - 1;
|
||||
while (pv >= 100) {
|
||||
memcpy(write_pointer - 1, &internal::decimal_table[(pv % 100) * 2], 2);
|
||||
@@ -519,8 +515,8 @@ simdjson_inline void string_builder::append(const T &opt) {
|
||||
|
||||
template <typename T>
|
||||
requires(require_custom_serialization<T>)
|
||||
simdjson_inline void string_builder::append(const T &val) {
|
||||
serialize(*this, val);
|
||||
simdjson_inline void string_builder::append(T &&val) {
|
||||
serialize(*this, std::forward<T>(val));
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
@@ -534,11 +530,11 @@ simdjson_inline void string_builder::append(const T &value) {
|
||||
#if SIMDJSON_SUPPORTS_RANGES && SIMDJSON_SUPPORTS_CONCEPTS
|
||||
// Support for range-based appending (std::ranges::view, etc.)
|
||||
template <std::ranges::range R>
|
||||
requires(!std::is_convertible<R, std::string_view>::value)
|
||||
requires(!std::is_convertible<R, std::string_view>::value && !require_custom_serialization<R>)
|
||||
simdjson_inline void string_builder::append(const R &range) noexcept {
|
||||
auto it = std::ranges::begin(range);
|
||||
auto end = std::ranges::end(range);
|
||||
if constexpr (concepts::is_pair<typename R::value_type>) {
|
||||
if constexpr (concepts::is_pair<std::ranges::range_value_t<R>>) {
|
||||
start_object();
|
||||
|
||||
if (it == end) {
|
||||
@@ -1,7 +1,3 @@
|
||||
/**
|
||||
* This file is part of the builder API. It is temporarily in the ondemand directory
|
||||
* but we will move it to a builder directory later.
|
||||
*/
|
||||
#ifndef SIMDJSON_GENERIC_STRING_BUILDER_H
|
||||
|
||||
#ifndef SIMDJSON_CONDITIONAL_INCLUDE
|
||||
@@ -24,9 +20,8 @@ struct has_custom_serialization : std::false_type {};
|
||||
|
||||
inline constexpr struct serialize_tag {
|
||||
template <typename T>
|
||||
requires custom_deserializable<T>
|
||||
constexpr void operator()(SIMDJSON_IMPLEMENTATION::builder::string_builder& b, T& obj) const{
|
||||
return tag_invoke(*this, b, obj);
|
||||
constexpr void operator()(SIMDJSON_IMPLEMENTATION::builder::string_builder& b, T&& obj) const{
|
||||
return tag_invoke(*this, b, std::forward<T>(obj));
|
||||
}
|
||||
|
||||
|
||||
@@ -165,7 +160,7 @@ public:
|
||||
|
||||
template <typename T>
|
||||
requires(require_custom_serialization<T>)
|
||||
simdjson_inline void append(const T &val);
|
||||
simdjson_inline void append(T &&val);
|
||||
|
||||
// Support for string-like types
|
||||
template <typename T>
|
||||
@@ -176,7 +171,7 @@ public:
|
||||
#if SIMDJSON_SUPPORTS_RANGES && SIMDJSON_SUPPORTS_CONCEPTS
|
||||
// Support for range-based appending (std::ranges::view, etc.)
|
||||
template <std::ranges::range R>
|
||||
requires (!std::is_convertible<R, std::string_view>::value)
|
||||
requires (!std::is_convertible<R, std::string_view>::value && !require_custom_serialization<R>)
|
||||
simdjson_inline void append(const R &range) noexcept;
|
||||
#endif
|
||||
/**
|
||||
@@ -301,4 +296,4 @@ simdjson_warn_unused simdjson_error to_json(const Z &z, std::string &s, size_t i
|
||||
|
||||
} // namespace simdjson
|
||||
|
||||
#endif // SIMDJSON_GENERIC_STRING_BUILDER_H
|
||||
#endif // SIMDJSON_GENERIC_STRING_BUILDER_H
|
||||
@@ -40,6 +40,7 @@ public:
|
||||
simdjson_warn_unused error_code stage1(const uint8_t *buf, size_t len, stage1_mode partial) noexcept final;
|
||||
simdjson_warn_unused error_code stage2(dom::document &doc) noexcept final;
|
||||
simdjson_warn_unused error_code stage2_next(dom::document &doc) noexcept final;
|
||||
simdjson_warn_unused std::pair<const uint8_t *,bool> parse_string_if_needed(const uint8_t *src, uint8_t *dst, bool allow_replacement) const noexcept final;
|
||||
simdjson_warn_unused uint8_t *parse_string(const uint8_t *src, uint8_t *dst, bool allow_replacement) const noexcept final;
|
||||
simdjson_warn_unused uint8_t *parse_wobbly_string(const uint8_t *src, uint8_t *dst) const noexcept final;
|
||||
inline simdjson_warn_unused error_code set_capacity(size_t capacity) noexcept final;
|
||||
|
||||
@@ -138,6 +138,9 @@ struct implementation_simdjson_result_base {
|
||||
*/
|
||||
simdjson_inline T&& value_unsafe() && noexcept;
|
||||
|
||||
using value_type = T;
|
||||
using error_type = error_code;
|
||||
|
||||
protected:
|
||||
/** users should never directly access first and second. **/
|
||||
T first{}; /** Users should never directly access 'first'. **/
|
||||
|
||||
@@ -167,7 +167,12 @@ simdjson_inline bool compute_float_64(int64_t power, uint64_t i, bool negative,
|
||||
// with a returned value of type value128 with a "low component" corresponding to the
|
||||
// 64-bit least significant bits of the product and with a "high component" corresponding
|
||||
// to the 64-bit most significant bits of the product.
|
||||
#if SIMDJSON_STATIC_REFLECTION
|
||||
simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index]);
|
||||
#else
|
||||
simdjson::internal::value128 firstproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index]);
|
||||
#endif
|
||||
|
||||
// Both i and power_of_five_128[index] have their most significant bit set to 1 which
|
||||
// implies that the either the most or the second most significant bit of the product
|
||||
// is 1. We pack values in this manner for efficiency reasons: it maximizes the use
|
||||
@@ -200,7 +205,11 @@ simdjson_inline bool compute_float_64(int64_t power, uint64_t i, bool negative,
|
||||
// with a returned value of type value128 with a "low component" corresponding to the
|
||||
// 64-bit least significant bits of the product and with a "high component" corresponding
|
||||
// to the 64-bit most significant bits of the product.
|
||||
#if SIMDJSON_STATIC_REFLECTION
|
||||
simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::powers_template<>::power_of_five_128[index + 1]);
|
||||
#else
|
||||
simdjson::internal::value128 secondproduct = full_multiplication(i, simdjson::internal::power_of_five_128[index + 1]);
|
||||
#endif
|
||||
firstproduct.low += secondproduct.high;
|
||||
if(secondproduct.high > firstproduct.low) { firstproduct.high++; }
|
||||
// As it has been proven by Noble Mushtak and Daniel Lemire in "Fast Number Parsing Without
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
#if defined(SIMDJSON_CONDITIONAL_INCLUDE) && !defined(SIMDJSON_GENERIC_ONDEMAND_DEPENDENCIES_H)
|
||||
#if defined(SIMDJSON_CONDITIONAL_INCLUDE) && !defined(SIMDJSON_GENERIC_BUILDER_DEPENDENCIES_H)
|
||||
#error simdjson/generic/ondemand/dependencies.h must be included before simdjson/generic/ondemand/amalgamated.h!
|
||||
#endif
|
||||
|
||||
@@ -14,9 +14,6 @@
|
||||
#include "simdjson/generic/ondemand/raw_json_string.h"
|
||||
#include "simdjson/generic/ondemand/parser.h"
|
||||
|
||||
// JSON builder - needed for extract_into functionality
|
||||
#include "simdjson/generic/ondemand/json_string_builder.h"
|
||||
|
||||
// All other declarations
|
||||
#include "simdjson/generic/ondemand/array.h"
|
||||
#include "simdjson/generic/ondemand/array_iterator.h"
|
||||
@@ -44,11 +41,10 @@
|
||||
#include "simdjson/generic/ondemand/object_iterator-inl.h"
|
||||
#include "simdjson/generic/ondemand/parser-inl.h"
|
||||
#include "simdjson/generic/ondemand/raw_json_string-inl.h"
|
||||
#include "simdjson/generic/ondemand/serialization-inl.h"
|
||||
#include "simdjson/generic/ondemand/token_iterator-inl.h"
|
||||
#include "simdjson/generic/ondemand/value_iterator-inl.h"
|
||||
#include "simdjson/generic/ondemand/serialization-inl.h"
|
||||
|
||||
// JSON builder inline definitions
|
||||
#include "simdjson/generic/ondemand/json_string_builder-inl.h"
|
||||
#include "simdjson/generic/ondemand/json_builder.h"
|
||||
// JSON path accessor (compile-time) - must be after inline definitions
|
||||
#include "simdjson/generic/ondemand/compile_time_accessors.h"
|
||||
|
||||
|
||||
@@ -170,6 +170,67 @@ inline simdjson_result<value> array::at_path(std::string_view json_path) noexcep
|
||||
return at_pointer(json_pointer);
|
||||
}
|
||||
|
||||
inline simdjson_result<std::vector<value>> array::at_path_with_wildcard(std::string_view json_path) noexcept {
|
||||
std::vector<value> result;
|
||||
|
||||
auto result_pair = get_next_key_and_json_path(json_path);
|
||||
std::string_view key = result_pair.first;
|
||||
std::string_view remaining_path = result_pair.second;
|
||||
// Wildcard case
|
||||
if(key=="*"){
|
||||
for(auto element: *this){
|
||||
|
||||
if(element.error()){
|
||||
return element.error();
|
||||
}
|
||||
|
||||
if(remaining_path.empty()){
|
||||
// Use value_unsafe() because we've already checked for errors above.
|
||||
// The 'element' is a simdjson_result<value> wrapper, and we need to extract
|
||||
// the underlying value. value_unsafe() is safe here because error() returned false.
|
||||
result.push_back(std::move(element).value_unsafe());
|
||||
|
||||
}else{
|
||||
auto nested_result = element.at_path_with_wildcard(remaining_path);
|
||||
|
||||
if(nested_result.error()){
|
||||
return nested_result.error();
|
||||
}
|
||||
// Same logic as above.
|
||||
std::vector<value> nested_matches = std::move(nested_result).value_unsafe();
|
||||
|
||||
result.insert(result.end(),
|
||||
std::make_move_iterator(nested_matches.begin()),
|
||||
std::make_move_iterator(nested_matches.end()));
|
||||
}
|
||||
}
|
||||
return result;
|
||||
}else{
|
||||
// Specific index case in which we access the element at the given index
|
||||
size_t idx=0;
|
||||
|
||||
for(char c:key){
|
||||
if(c < '0' || c > '9'){
|
||||
return INVALID_JSON_POINTER;
|
||||
}
|
||||
idx = idx*10 + (c - '0');
|
||||
}
|
||||
|
||||
auto element = at(idx);
|
||||
|
||||
if(element.error()){
|
||||
return element.error();
|
||||
}
|
||||
|
||||
if(remaining_path.empty()){
|
||||
result.push_back(std::move(element).value_unsafe());
|
||||
return result;
|
||||
}else{
|
||||
return element.at_path_with_wildcard(remaining_path);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
simdjson_inline simdjson_result<value> array::at(size_t index) noexcept {
|
||||
size_t i = 0;
|
||||
for (auto value : *this) {
|
||||
@@ -228,6 +289,10 @@ simdjson_inline simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::value> simdj
|
||||
if (error()) { return error(); }
|
||||
return first.at_path(json_path);
|
||||
}
|
||||
simdjson_inline simdjson_result<std::vector<SIMDJSON_IMPLEMENTATION::ondemand::value>> simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::array>::at_path_with_wildcard(std::string_view json_path) noexcept {
|
||||
if (error()) { return error(); }
|
||||
return first.at_path_with_wildcard(json_path);
|
||||
}
|
||||
simdjson_inline simdjson_result<std::string_view> simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::array>::raw_json() noexcept {
|
||||
if (error()) { return error(); }
|
||||
return first.raw_json();
|
||||
|
||||
@@ -5,6 +5,7 @@
|
||||
#include "simdjson/generic/ondemand/base.h"
|
||||
#include "simdjson/generic/implementation_simdjson_result_base.h"
|
||||
#include "simdjson/generic/ondemand/value_iterator.h"
|
||||
#include <vector>
|
||||
#endif // SIMDJSON_CONDITIONAL_INCLUDE
|
||||
|
||||
namespace simdjson {
|
||||
@@ -107,7 +108,7 @@ public:
|
||||
* JSONPath queries that trivially convertible to JSON Pointer queries: key
|
||||
* names and array indices.
|
||||
*
|
||||
* https://datatracker.ietf.org/doc/html/draft-normington-jsonpath-00
|
||||
* https://www.rfc-editor.org/rfc/rfc9535 (RFC 9535)
|
||||
*
|
||||
* @return The value associated with the given JSONPath expression, or:
|
||||
* - INVALID_JSON_POINTER if the JSONPath to JSON Pointer conversion fails
|
||||
@@ -117,6 +118,15 @@ public:
|
||||
*/
|
||||
inline simdjson_result<value> at_path(std::string_view json_path) noexcept;
|
||||
|
||||
/**
|
||||
* Get all values matching the given JSONPath expression with wildcard support.
|
||||
* Supports wildcard patterns like "[*]" to match all array elements.
|
||||
*
|
||||
* @param json_path JSONPath expression with wildcards
|
||||
* @return Vector of values matching the wildcard pattern
|
||||
*/
|
||||
inline simdjson_result<std::vector<value>> at_path_with_wildcard(std::string_view json_path) noexcept;
|
||||
|
||||
/**
|
||||
* Consumes the array and returns a string_view instance corresponding to the
|
||||
* array as represented in JSON. It points inside the original document.
|
||||
@@ -239,6 +249,7 @@ public:
|
||||
simdjson_inline simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::value> at(size_t index) noexcept;
|
||||
simdjson_inline simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::value> at_pointer(std::string_view json_pointer) noexcept;
|
||||
simdjson_inline simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::value> at_path(std::string_view json_path) noexcept;
|
||||
simdjson_inline simdjson_result<std::vector<SIMDJSON_IMPLEMENTATION::ondemand::value>> at_path_with_wildcard(std::string_view json_path) noexcept;
|
||||
simdjson_inline simdjson_result<std::string_view> raw_json() noexcept;
|
||||
#if SIMDJSON_SUPPORTS_CONCEPTS
|
||||
// TODO: move this code into object-inl.h
|
||||
|
||||
@@ -17,6 +17,10 @@ simdjson_inline array_iterator::array_iterator(const value_iterator &_iter) noex
|
||||
{}
|
||||
|
||||
simdjson_inline simdjson_result<value> array_iterator::operator*() noexcept {
|
||||
#if SIMDJSON_DEVELOPMENT_CHECKS
|
||||
SIMDJSON_ASSUME(!has_been_referenced);
|
||||
has_been_referenced = true;
|
||||
#endif
|
||||
if (iter.error()) { iter.abandon(); return iter.error(); }
|
||||
return value(iter.child());
|
||||
}
|
||||
@@ -27,6 +31,9 @@ simdjson_inline bool array_iterator::operator!=(const array_iterator &) const no
|
||||
return iter.is_open();
|
||||
}
|
||||
simdjson_inline array_iterator &array_iterator::operator++() noexcept {
|
||||
#if SIMDJSON_DEVELOPMENT_CHECKS
|
||||
has_been_referenced = false;
|
||||
#endif
|
||||
error_code error;
|
||||
// PERF NOTE this is a safety rail ... users should exit loops as soon as they receive an error, so we'll never get here.
|
||||
// However, it does not seem to make a perf difference, so we add it out of an abundance of caution.
|
||||
|
||||
@@ -2,6 +2,7 @@
|
||||
|
||||
#ifndef SIMDJSON_CONDITIONAL_INCLUDE
|
||||
#define SIMDJSON_GENERIC_ONDEMAND_ARRAY_ITERATOR_H
|
||||
#include <iterator>
|
||||
#include "simdjson/generic/implementation_simdjson_result_base.h"
|
||||
#include "simdjson/generic/ondemand/base.h"
|
||||
#include "simdjson/generic/ondemand/value_iterator.h"
|
||||
@@ -17,11 +18,17 @@ namespace ondemand {
|
||||
*
|
||||
* This is an input_iterator, meaning:
|
||||
* - It is forward-only
|
||||
* - * must be called exactly once per element.
|
||||
* - * must be called at most once per element.
|
||||
* - ++ must be called exactly once in between each * (*, ++, *, ++, * ...)
|
||||
*/
|
||||
class array_iterator {
|
||||
public:
|
||||
using iterator_category = std::input_iterator_tag;
|
||||
using value_type = simdjson_result<value>;
|
||||
using difference_type = std::ptrdiff_t;
|
||||
using pointer = void;
|
||||
using reference = value_type;
|
||||
|
||||
/** Create a new, invalid array iterator. */
|
||||
simdjson_inline array_iterator() noexcept = default;
|
||||
|
||||
@@ -65,6 +72,9 @@ public:
|
||||
simdjson_warn_unused simdjson_inline bool at_end() const noexcept;
|
||||
|
||||
private:
|
||||
#if SIMDJSON_DEVELOPMENT_CHECKS
|
||||
bool has_been_referenced{false};
|
||||
#endif
|
||||
value_iterator iter{};
|
||||
|
||||
simdjson_inline array_iterator(const value_iterator &iter) noexcept;
|
||||
@@ -82,6 +92,12 @@ namespace simdjson {
|
||||
|
||||
template<>
|
||||
struct simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::array_iterator> : public SIMDJSON_IMPLEMENTATION::implementation_simdjson_result_base<SIMDJSON_IMPLEMENTATION::ondemand::array_iterator> {
|
||||
using iterator_category = std::input_iterator_tag;
|
||||
using value_type = simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::value>;
|
||||
using difference_type = std::ptrdiff_t;
|
||||
using pointer = void;
|
||||
using reference = value_type;
|
||||
|
||||
simdjson_inline simdjson_result(SIMDJSON_IMPLEMENTATION::ondemand::array_iterator &&value) noexcept; ///< @private
|
||||
simdjson_inline simdjson_result(error_code error) noexcept; ///< @private
|
||||
simdjson_inline simdjson_result() noexcept = default;
|
||||
|
||||
@@ -0,0 +1,938 @@
|
||||
/**
|
||||
* Compile-time JSON Path and JSON Pointer accessors using C++26 reflection (P2996)
|
||||
*
|
||||
* This file validates JSON paths/pointers against struct definitions at compile time
|
||||
* and generates optimized accessor code with zero runtime overhead.
|
||||
*
|
||||
* ## How It Works
|
||||
*
|
||||
* **Compile Time**: Path is parsed, validated against struct, types are checked
|
||||
* **Runtime**: Direct navigation with no parsing or validation overhead
|
||||
*
|
||||
* Example:
|
||||
* ```cpp
|
||||
* struct User { std::string name; std::vector<std::string> emails; };
|
||||
*
|
||||
* std::string email;
|
||||
* path_accessor<User, ".emails[0]">::extract_field(doc, email);
|
||||
*
|
||||
* // Compile time validates:
|
||||
* // 1. User has "emails" field
|
||||
* // 2. "emails" is array-like
|
||||
* // 3. Element type is std::string
|
||||
* // 4. static_assert(^^std::string == ^^std::string)
|
||||
*
|
||||
* // Runtime just navigates:
|
||||
* // doc.get_object().find_field("emails").get_array().at(0).get(email)
|
||||
* ```
|
||||
*
|
||||
* ## Key Reflection APIs
|
||||
*
|
||||
* - `^^Type`: Reflect operator, converts type to std::meta::info
|
||||
* - `std::meta::nonstatic_data_members_of(type)`: Get all fields of a struct
|
||||
* - `std::meta::identifier_of(member)`: Get field name as string_view
|
||||
* - `std::meta::type_of(member)`: Get reflected type of a field
|
||||
* - `std::meta::is_array_type(type)`: Check if C-style array
|
||||
* - `std::meta::remove_extent(array)`: Extract element type from array
|
||||
* - `std::meta::members_of(type)`: Get all members including typedefs
|
||||
* - `std::meta::is_type(member)`: Check if member is a type (vs field)
|
||||
*
|
||||
* All operations execute at compile time in consteval contexts.
|
||||
*/
|
||||
#ifndef SIMDJSON_GENERIC_ONDEMAND_COMPILE_TIME_ACCESSORS_H
|
||||
|
||||
#ifndef SIMDJSON_CONDITIONAL_INCLUDE
|
||||
#define SIMDJSON_GENERIC_ONDEMAND_COMPILE_TIME_ACCESSORS_H
|
||||
|
||||
#endif // SIMDJSON_CONDITIONAL_INCLUDE
|
||||
|
||||
// Arguably, we should just check SIMDJSON_STATIC_REFLECTION since it
|
||||
// is unlikely that we will have reflection support without concepts support.
|
||||
#if SIMDJSON_SUPPORTS_CONCEPTS && SIMDJSON_STATIC_REFLECTION
|
||||
|
||||
#include <string_view>
|
||||
#include <cstddef>
|
||||
#include <array>
|
||||
|
||||
namespace simdjson {
|
||||
namespace SIMDJSON_IMPLEMENTATION {
|
||||
namespace ondemand {
|
||||
/***
|
||||
* JSONPath implementation for compile-time access
|
||||
* RFC 9535 JSONPath: Query Expressions for JSON, https://www.rfc-editor.org/rfc/rfc9535
|
||||
*/
|
||||
namespace json_path {
|
||||
|
||||
// Note: value type must be fully defined before this header is included
|
||||
// This is ensured by including this in amalgamated.h after value-inl.h
|
||||
|
||||
using ::simdjson::SIMDJSON_IMPLEMENTATION::ondemand::value;
|
||||
|
||||
// Path step types
|
||||
enum class step_type {
|
||||
field, // .field_name or ["field_name"]
|
||||
array_index // [index]
|
||||
};
|
||||
|
||||
// Represents a single step in a JSON path
|
||||
template<std::size_t N>
|
||||
struct path_step {
|
||||
step_type type;
|
||||
char key[N]; // Field name (empty for array indices)
|
||||
std::size_t index; // Array index (0 for field access)
|
||||
|
||||
constexpr path_step(step_type t, const char (&k)[N], std::size_t idx = 0)
|
||||
: type(t), index(idx) {
|
||||
for (std::size_t i = 0; i < N; ++i) {
|
||||
key[i] = k[i];
|
||||
}
|
||||
}
|
||||
|
||||
constexpr std::string_view key_view() const {
|
||||
return {key, N - 1};
|
||||
}
|
||||
};
|
||||
|
||||
// Helper to create field step
|
||||
template<std::size_t N>
|
||||
consteval auto make_field_step(const char (&name)[N]) {
|
||||
return path_step<N>(step_type::field, name, 0);
|
||||
}
|
||||
|
||||
// Helper to create array index step
|
||||
consteval auto make_index_step(std::size_t idx) {
|
||||
return path_step<1>(step_type::array_index, "", idx);
|
||||
}
|
||||
|
||||
// Parse state for compile-time JSON path parsing
|
||||
struct parse_result {
|
||||
bool success;
|
||||
std::size_t pos;
|
||||
std::string_view error_msg;
|
||||
};
|
||||
|
||||
// Compile-time JSON path parser
|
||||
// Supports subset: .field, ["field"], [index], nested combinations
|
||||
template<constevalutil::fixed_string Path>
|
||||
struct json_path_parser {
|
||||
static constexpr std::string_view path_str = Path.view();
|
||||
|
||||
// Skip leading $ if present
|
||||
static consteval std::size_t skip_root() {
|
||||
if (!path_str.empty() && path_str[0] == '$') {
|
||||
return 1;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
// Count the number of steps in the path at compile time
|
||||
static consteval std::size_t count_steps() {
|
||||
std::size_t count = 0;
|
||||
std::size_t i = skip_root();
|
||||
|
||||
while (i < path_str.size()) {
|
||||
if (path_str[i] == '.') {
|
||||
// Field access: .field
|
||||
++i;
|
||||
if (i >= path_str.size()) break;
|
||||
|
||||
// Skip field name
|
||||
while (i < path_str.size() && path_str[i] != '.' && path_str[i] != '[') {
|
||||
++i;
|
||||
}
|
||||
++count;
|
||||
} else if (path_str[i] == '[') {
|
||||
// Array or bracket notation
|
||||
++i;
|
||||
if (i >= path_str.size()) break;
|
||||
|
||||
if (path_str[i] == '"' || path_str[i] == '\'') {
|
||||
// Field access: ["field"] or ['field']
|
||||
char quote = path_str[i];
|
||||
++i;
|
||||
while (i < path_str.size() && path_str[i] != quote) {
|
||||
++i;
|
||||
}
|
||||
if (i < path_str.size()) ++i; // skip closing quote
|
||||
if (i < path_str.size() && path_str[i] == ']') ++i;
|
||||
} else {
|
||||
// Array index: [0], [123]
|
||||
while (i < path_str.size() && path_str[i] != ']') {
|
||||
++i;
|
||||
}
|
||||
if (i < path_str.size()) ++i; // skip ]
|
||||
}
|
||||
++count;
|
||||
} else {
|
||||
++i;
|
||||
}
|
||||
}
|
||||
|
||||
return count;
|
||||
}
|
||||
|
||||
// Parse a field name at compile time
|
||||
static consteval std::size_t parse_field_name(std::size_t start, char* out, std::size_t max_len) {
|
||||
std::size_t len = 0;
|
||||
std::size_t i = start;
|
||||
|
||||
while (i < path_str.size() && path_str[i] != '.' && path_str[i] != '[' && len < max_len - 1) {
|
||||
out[len++] = path_str[i++];
|
||||
}
|
||||
out[len] = '\0';
|
||||
return i;
|
||||
}
|
||||
|
||||
// Parse an array index at compile time
|
||||
static consteval std::pair<std::size_t, std::size_t> parse_array_index(std::size_t start) {
|
||||
std::size_t index = 0;
|
||||
std::size_t i = start;
|
||||
|
||||
while (i < path_str.size() && path_str[i] >= '0' && path_str[i] <= '9') {
|
||||
index = index * 10 + (path_str[i] - '0');
|
||||
++i;
|
||||
}
|
||||
|
||||
return {i, index};
|
||||
}
|
||||
};
|
||||
|
||||
// Compile-time path accessor generator
|
||||
template<typename T, constevalutil::fixed_string Path>
|
||||
struct path_accessor {
|
||||
using value = ::simdjson::SIMDJSON_IMPLEMENTATION::ondemand::value;
|
||||
|
||||
static constexpr auto parser = json_path_parser<Path>();
|
||||
static constexpr std::size_t num_steps = parser.count_steps();
|
||||
static constexpr std::string_view path_view = Path.view();
|
||||
|
||||
// Compile-time accessor generation
|
||||
// If T is a struct, validates the path at compile time
|
||||
// If T is void, skips validation
|
||||
template<typename DocOrValue>
|
||||
static inline simdjson_result<value> access(DocOrValue& doc_or_val) noexcept {
|
||||
// Validate path at compile time if T is a struct
|
||||
if constexpr (std::is_class_v<T>) {
|
||||
constexpr bool path_valid = validate_path();
|
||||
static_assert(path_valid, "JSON path does not match struct definition");
|
||||
}
|
||||
|
||||
// Parse the path at compile time to build access steps
|
||||
return access_impl<parser.skip_root()>(doc_or_val.get_value());
|
||||
}
|
||||
|
||||
// Extract value at path directly into target with compile-time type validation
|
||||
// Example: std::string name; path_accessor<User, ".name">::extract_field(doc, name);
|
||||
template<typename DocOrValue, typename FieldType>
|
||||
static inline error_code extract_field(DocOrValue& doc_or_val, FieldType& target) noexcept {
|
||||
static_assert(std::is_class_v<T>, "extract_field requires T to be a struct type for validation");
|
||||
|
||||
// Validate path exists in struct definition
|
||||
constexpr bool path_valid = validate_path();
|
||||
static_assert(path_valid, "JSON path does not match struct definition");
|
||||
|
||||
// Get the type at the end of the path
|
||||
constexpr auto final_type = get_final_type();
|
||||
|
||||
// Verify target type matches the field type
|
||||
static_assert(final_type == ^^FieldType, "Target type does not match the field type at the path");
|
||||
|
||||
// All validation done at compile time - just navigate and extract
|
||||
auto json_value = access_impl<parser.skip_root()>(doc_or_val.get_value());
|
||||
if (json_value.error()) return json_value.error();
|
||||
|
||||
return json_value.get(target);
|
||||
}
|
||||
|
||||
private:
|
||||
// Get the final type by walking the path through the struct type
|
||||
template<typename U = T>
|
||||
static consteval std::enable_if_t<std::is_class_v<U>, std::meta::info> get_final_type() {
|
||||
auto current_type = ^^T;
|
||||
std::size_t i = parser.skip_root();
|
||||
|
||||
while (i < path_view.size()) {
|
||||
if (path_view[i] == '.') {
|
||||
// .field syntax
|
||||
++i;
|
||||
std::size_t field_start = i;
|
||||
while (i < path_view.size() && path_view[i] != '.' && path_view[i] != '[') {
|
||||
++i;
|
||||
}
|
||||
|
||||
std::string_view field_name = path_view.substr(field_start, i - field_start);
|
||||
|
||||
auto members = std::meta::nonstatic_data_members_of(
|
||||
current_type, std::meta::access_context::unchecked()
|
||||
);
|
||||
|
||||
for (auto mem : members) {
|
||||
if (std::meta::identifier_of(mem) == field_name) {
|
||||
current_type = std::meta::type_of(mem);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
} else if (path_view[i] == '[') {
|
||||
++i;
|
||||
if (i >= path_view.size()) break;
|
||||
|
||||
if (path_view[i] == '"' || path_view[i] == '\'') {
|
||||
// ["field"] syntax
|
||||
char quote = path_view[i];
|
||||
++i;
|
||||
std::size_t field_start = i;
|
||||
while (i < path_view.size() && path_view[i] != quote) {
|
||||
++i;
|
||||
}
|
||||
|
||||
std::string_view field_name = path_view.substr(field_start, i - field_start);
|
||||
if (i < path_view.size()) ++i; // skip quote
|
||||
if (i < path_view.size() && path_view[i] == ']') ++i;
|
||||
|
||||
auto members = std::meta::nonstatic_data_members_of(
|
||||
current_type, std::meta::access_context::unchecked()
|
||||
);
|
||||
|
||||
for (auto mem : members) {
|
||||
if (std::meta::identifier_of(mem) == field_name) {
|
||||
current_type = std::meta::type_of(mem);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
} else {
|
||||
// [index] syntax - extract element type
|
||||
while (i < path_view.size() && path_view[i] >= '0' && path_view[i] <= '9') {
|
||||
++i;
|
||||
}
|
||||
if (i < path_view.size() && path_view[i] == ']') ++i;
|
||||
|
||||
current_type = get_element_type_reflected(current_type);
|
||||
}
|
||||
} else {
|
||||
++i;
|
||||
}
|
||||
}
|
||||
|
||||
return current_type;
|
||||
}
|
||||
|
||||
private:
|
||||
// Walk path and extract directly into final field using compile-time reflection
|
||||
template<std::meta::info CurrentType, std::size_t PathPos, typename TargetType>
|
||||
static inline error_code extract_with_reflection(simdjson_result<value> current, TargetType& target_ref) noexcept {
|
||||
if (current.error()) return current.error();
|
||||
|
||||
// Base case: end of path - extract into target
|
||||
if constexpr (PathPos >= path_view.size()) {
|
||||
return current.get(target_ref);
|
||||
}
|
||||
// Field access: .field_name
|
||||
else if constexpr (path_view[PathPos] == '.') {
|
||||
constexpr auto field_info = parse_next_field(PathPos);
|
||||
constexpr std::string_view field_name = std::get<0>(field_info);
|
||||
constexpr std::size_t next_pos = std::get<1>(field_info);
|
||||
|
||||
constexpr auto member_info = find_member_by_name(CurrentType, field_name);
|
||||
static_assert(member_info != ^^void, "Field not found in struct");
|
||||
|
||||
constexpr auto member_type = std::meta::type_of(member_info);
|
||||
|
||||
auto obj_result = current.get_object();
|
||||
if (obj_result.error()) return obj_result.error();
|
||||
auto obj = obj_result.value_unsafe();
|
||||
auto field_value = obj.find_field_unordered(field_name);
|
||||
|
||||
if constexpr (next_pos >= path_view.size()) {
|
||||
return field_value.get(target_ref);
|
||||
} else {
|
||||
return extract_with_reflection<member_type, next_pos>(field_value, target_ref);
|
||||
}
|
||||
}
|
||||
// Bracket notation: [index] or ["field"]
|
||||
else if constexpr (path_view[PathPos] == '[') {
|
||||
constexpr auto bracket_info = parse_bracket(PathPos);
|
||||
constexpr bool is_field = std::get<0>(bracket_info);
|
||||
constexpr std::size_t next_pos = std::get<2>(bracket_info);
|
||||
|
||||
if constexpr (is_field) {
|
||||
constexpr std::string_view field_name = std::get<1>(bracket_info);
|
||||
constexpr auto member_info = find_member_by_name(CurrentType, field_name);
|
||||
static_assert(member_info != ^^void, "Field not found in struct");
|
||||
constexpr auto member_type = std::meta::type_of(member_info);
|
||||
|
||||
auto obj_result = current.get_object();
|
||||
if (obj_result.error()) return obj_result.error();
|
||||
auto obj = obj_result.value_unsafe();
|
||||
auto field_value = obj.find_field_unordered(field_name);
|
||||
|
||||
if constexpr (next_pos >= path_view.size()) {
|
||||
return field_value.get(target_ref);
|
||||
} else {
|
||||
return extract_with_reflection<member_type, next_pos>(field_value, target_ref);
|
||||
}
|
||||
} else {
|
||||
constexpr std::size_t index = std::get<3>(bracket_info);
|
||||
constexpr auto elem_type = get_element_type_reflected(CurrentType);
|
||||
static_assert(elem_type != ^^void, "Could not determine array element type");
|
||||
|
||||
auto arr_result = current.get_array();
|
||||
if (arr_result.error()) return arr_result.error();
|
||||
auto arr = arr_result.value_unsafe();
|
||||
auto elem_value = arr.at(index);
|
||||
|
||||
if constexpr (next_pos >= path_view.size()) {
|
||||
return elem_value.get(target_ref);
|
||||
} else {
|
||||
return extract_with_reflection<elem_type, next_pos>(elem_value, target_ref);
|
||||
}
|
||||
}
|
||||
}
|
||||
// Skip unexpected characters and continue
|
||||
else {
|
||||
return extract_with_reflection<CurrentType, PathPos + 1>(current, target_ref);
|
||||
}
|
||||
}
|
||||
|
||||
// Find member by name in reflected type
|
||||
static consteval std::meta::info find_member_by_name(std::meta::info type_refl, std::string_view name) {
|
||||
auto members = std::meta::nonstatic_data_members_of(type_refl, std::meta::access_context::unchecked());
|
||||
for (auto mem : members) {
|
||||
if (std::meta::identifier_of(mem) == name) {
|
||||
return mem;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Generate compile-time accessor code by walking the path
|
||||
template<std::size_t PathPos>
|
||||
static inline simdjson_result<value> access_impl(simdjson_result<value> current) noexcept {
|
||||
if (current.error()) return current;
|
||||
|
||||
if constexpr (PathPos >= path_view.size()) {
|
||||
return current;
|
||||
} else if constexpr (path_view[PathPos] == '.') {
|
||||
constexpr auto field_info = parse_next_field(PathPos);
|
||||
constexpr std::string_view field_name = std::get<0>(field_info);
|
||||
constexpr std::size_t next_pos = std::get<1>(field_info);
|
||||
|
||||
auto obj_result = current.get_object();
|
||||
if (obj_result.error()) return obj_result.error();
|
||||
|
||||
auto obj = obj_result.value_unsafe();
|
||||
auto next_value = obj.find_field_unordered(field_name);
|
||||
|
||||
return access_impl<next_pos>(next_value);
|
||||
|
||||
} else if constexpr (path_view[PathPos] == '[') {
|
||||
constexpr auto bracket_info = parse_bracket(PathPos);
|
||||
constexpr bool is_field = std::get<0>(bracket_info);
|
||||
constexpr std::size_t next_pos = std::get<2>(bracket_info);
|
||||
|
||||
if constexpr (is_field) {
|
||||
constexpr std::string_view field_name = std::get<1>(bracket_info);
|
||||
|
||||
auto obj_result = current.get_object();
|
||||
if (obj_result.error()) return obj_result.error();
|
||||
|
||||
auto obj = obj_result.value_unsafe();
|
||||
auto next_value = obj.find_field_unordered(field_name);
|
||||
|
||||
return access_impl<next_pos>(next_value);
|
||||
|
||||
} else {
|
||||
constexpr std::size_t index = std::get<3>(bracket_info);
|
||||
|
||||
auto arr_result = current.get_array();
|
||||
if (arr_result.error()) return arr_result.error();
|
||||
|
||||
auto arr = arr_result.value_unsafe();
|
||||
auto next_value = arr.at(index);
|
||||
|
||||
return access_impl<next_pos>(next_value);
|
||||
}
|
||||
} else {
|
||||
return access_impl<PathPos + 1>(current);
|
||||
}
|
||||
}
|
||||
|
||||
// Parse next field name
|
||||
static consteval auto parse_next_field(std::size_t start) {
|
||||
std::size_t i = start + 1;
|
||||
std::size_t field_start = i;
|
||||
while (i < path_view.size() && path_view[i] != '.' && path_view[i] != '[') {
|
||||
++i;
|
||||
}
|
||||
std::string_view field_name = path_view.substr(field_start, i - field_start);
|
||||
return std::make_tuple(field_name, i);
|
||||
}
|
||||
|
||||
// Parse bracket notation: returns (is_field, field_name, next_pos, index)
|
||||
static consteval auto parse_bracket(std::size_t start) {
|
||||
std::size_t i = start + 1; // skip '['
|
||||
|
||||
if (i < path_view.size() && (path_view[i] == '"' || path_view[i] == '\'')) {
|
||||
// Field access
|
||||
char quote = path_view[i];
|
||||
++i;
|
||||
std::size_t field_start = i;
|
||||
while (i < path_view.size() && path_view[i] != quote) {
|
||||
++i;
|
||||
}
|
||||
std::string_view field_name = path_view.substr(field_start, i - field_start);
|
||||
if (i < path_view.size()) ++i; // skip closing quote
|
||||
if (i < path_view.size() && path_view[i] == ']') ++i;
|
||||
|
||||
return std::make_tuple(true, field_name, i, std::size_t(0));
|
||||
} else {
|
||||
// Array index
|
||||
std::size_t index = 0;
|
||||
while (i < path_view.size() && path_view[i] >= '0' && path_view[i] <= '9') {
|
||||
index = index * 10 + (path_view[i] - '0');
|
||||
++i;
|
||||
}
|
||||
if (i < path_view.size() && path_view[i] == ']') ++i;
|
||||
|
||||
return std::make_tuple(false, std::string_view{}, i, index);
|
||||
}
|
||||
}
|
||||
|
||||
public:
|
||||
// Check if reflected type is array-like (C-style array or indexable container)
|
||||
// Uses reflection to test: 1) std::meta::is_array_type() for C arrays
|
||||
// 2) std::meta::substitute() to test concepts::indexable_container concept
|
||||
static consteval bool is_array_like_reflected(std::meta::info type_reflection) {
|
||||
if (std::meta::is_array_type(type_reflection)) {
|
||||
return true;
|
||||
}
|
||||
|
||||
if (std::meta::can_substitute(^^concepts::indexable_container_v, {type_reflection})) {
|
||||
return std::meta::extract<bool>(std::meta::substitute(^^concepts::indexable_container_v, {type_reflection}));
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
// Extract element type from reflected array or container
|
||||
// For C arrays: uses std::meta::remove_extent()
|
||||
// For containers: finds value_type member using std::meta::members_of()
|
||||
static consteval std::meta::info get_element_type_reflected(std::meta::info type_reflection) {
|
||||
if (std::meta::is_array_type(type_reflection)) {
|
||||
return std::meta::remove_extent(type_reflection);
|
||||
}
|
||||
|
||||
auto members = std::meta::members_of(type_reflection, std::meta::access_context::unchecked());
|
||||
for (auto mem : members) {
|
||||
if (std::meta::is_type(mem)) {
|
||||
auto name = std::meta::identifier_of(mem);
|
||||
if (name == "value_type") {
|
||||
return mem;
|
||||
}
|
||||
}
|
||||
}
|
||||
return ^^void;
|
||||
}
|
||||
|
||||
private:
|
||||
// Check if type has member with given name
|
||||
template<typename Type>
|
||||
static consteval bool has_member(std::string_view member_name) {
|
||||
constexpr auto members = std::meta::nonstatic_data_members_of(^^Type, std::meta::access_context::unchecked());
|
||||
for (auto mem : members) {
|
||||
if (std::meta::identifier_of(mem) == member_name) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
// Get type of member by name
|
||||
template<typename Type>
|
||||
static consteval auto get_member_type(std::string_view member_name) {
|
||||
constexpr auto members = std::meta::nonstatic_data_members_of(^^Type, std::meta::access_context::unchecked());
|
||||
for (auto mem : members) {
|
||||
if (std::meta::identifier_of(mem) == member_name) {
|
||||
return std::meta::type_of(mem);
|
||||
}
|
||||
}
|
||||
return ^^void;
|
||||
}
|
||||
|
||||
// Check if non-reflected type is array-like
|
||||
template<typename Type>
|
||||
static consteval bool is_container_type() {
|
||||
using BaseType = std::remove_cvref_t<Type>;
|
||||
if constexpr (requires { typename BaseType::value_type; }) {
|
||||
return true;
|
||||
}
|
||||
if constexpr (std::is_array_v<BaseType>) {
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
// Extract element type from non-reflected container
|
||||
template<typename Type>
|
||||
using extract_element_type = std::conditional_t<
|
||||
requires { typename std::remove_cvref_t<Type>::value_type; },
|
||||
typename std::remove_cvref_t<Type>::value_type,
|
||||
std::conditional_t<
|
||||
std::is_array_v<std::remove_cvref_t<Type>>,
|
||||
std::remove_extent_t<std::remove_cvref_t<Type>>,
|
||||
void
|
||||
>
|
||||
>;
|
||||
|
||||
// Validate path matches struct definition
|
||||
static consteval bool validate_path() {
|
||||
if constexpr (!std::is_class_v<T>) {
|
||||
return true;
|
||||
}
|
||||
|
||||
auto current_type = ^^T;
|
||||
std::size_t i = parser.skip_root();
|
||||
|
||||
while (i < path_view.size()) {
|
||||
if (path_view[i] == '.') {
|
||||
++i;
|
||||
std::size_t field_start = i;
|
||||
while (i < path_view.size() && path_view[i] != '.' && path_view[i] != '[') {
|
||||
++i;
|
||||
}
|
||||
|
||||
std::string_view field_name = path_view.substr(field_start, i - field_start);
|
||||
|
||||
bool found = false;
|
||||
auto members = std::meta::nonstatic_data_members_of(current_type, std::meta::access_context::unchecked());
|
||||
|
||||
for (auto mem : members) {
|
||||
if (std::meta::identifier_of(mem) == field_name) {
|
||||
current_type = std::meta::type_of(mem);
|
||||
found = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if (!found) {
|
||||
return false;
|
||||
}
|
||||
|
||||
} else if (path_view[i] == '[') {
|
||||
++i;
|
||||
if (i >= path_view.size()) return false;
|
||||
|
||||
if (path_view[i] == '"' || path_view[i] == '\'') {
|
||||
char quote = path_view[i];
|
||||
++i;
|
||||
std::size_t field_start = i;
|
||||
while (i < path_view.size() && path_view[i] != quote) {
|
||||
++i;
|
||||
}
|
||||
|
||||
std::string_view field_name = path_view.substr(field_start, i - field_start);
|
||||
if (i < path_view.size()) ++i;
|
||||
if (i < path_view.size() && path_view[i] == ']') ++i;
|
||||
|
||||
bool found = false;
|
||||
auto members = std::meta::nonstatic_data_members_of(current_type, std::meta::access_context::unchecked());
|
||||
|
||||
for (auto mem : members) {
|
||||
if (std::meta::identifier_of(mem) == field_name) {
|
||||
current_type = std::meta::type_of(mem);
|
||||
found = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if (!found) {
|
||||
return false;
|
||||
}
|
||||
|
||||
} else {
|
||||
while (i < path_view.size() && path_view[i] >= '0' && path_view[i] <= '9') {
|
||||
++i;
|
||||
}
|
||||
|
||||
if (i < path_view.size() && path_view[i] == ']') ++i;
|
||||
|
||||
if (!is_array_like_reflected(current_type)) {
|
||||
return false;
|
||||
}
|
||||
|
||||
auto new_type = get_element_type_reflected(current_type);
|
||||
|
||||
if (new_type == ^^void) {
|
||||
return false;
|
||||
}
|
||||
|
||||
current_type = new_type;
|
||||
}
|
||||
} else {
|
||||
++i;
|
||||
}
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
};
|
||||
|
||||
// Compile-time path accessor with validation
|
||||
template<typename T, constevalutil::fixed_string Path, typename DocOrValue>
|
||||
inline simdjson_result<::simdjson::SIMDJSON_IMPLEMENTATION::ondemand::value> at_path_compiled(DocOrValue& doc_or_val) noexcept {
|
||||
using accessor = path_accessor<T, Path>;
|
||||
return accessor::access(doc_or_val);
|
||||
}
|
||||
|
||||
// Overload without type parameter (no validation)
|
||||
template<constevalutil::fixed_string Path, typename DocOrValue>
|
||||
inline simdjson_result<::simdjson::SIMDJSON_IMPLEMENTATION::ondemand::value> at_path_compiled(DocOrValue& doc_or_val) noexcept {
|
||||
using accessor = path_accessor<void, Path>;
|
||||
return accessor::access(doc_or_val);
|
||||
}
|
||||
|
||||
// ============================================================================
|
||||
// JSON Pointer Compile-Time Support (RFC 6901)
|
||||
// ============================================================================
|
||||
|
||||
// JSON Pointer parser: /field/0/nested (slash-separated)
|
||||
template<constevalutil::fixed_string Pointer>
|
||||
struct json_pointer_parser {
|
||||
static constexpr std::string_view pointer_str = Pointer.view();
|
||||
|
||||
// Unescape token: ~0 -> ~, ~1 -> /
|
||||
static consteval void unescape_token(std::string_view src, char* dest, std::size_t& out_len) {
|
||||
out_len = 0;
|
||||
for (std::size_t i = 0; i < src.size(); ++i) {
|
||||
if (src[i] == '~' && i + 1 < src.size()) {
|
||||
if (src[i + 1] == '0') {
|
||||
dest[out_len++] = '~';
|
||||
++i;
|
||||
} else if (src[i + 1] == '1') {
|
||||
dest[out_len++] = '/';
|
||||
++i;
|
||||
} else {
|
||||
dest[out_len++] = src[i];
|
||||
}
|
||||
} else {
|
||||
dest[out_len++] = src[i];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Check if token is numeric
|
||||
static consteval bool is_numeric(std::string_view token) {
|
||||
if (token.empty()) return false;
|
||||
if (token[0] == '0' && token.size() > 1) return false;
|
||||
for (char c : token) {
|
||||
if (c < '0' || c > '9') return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
// Parse numeric token to index
|
||||
static consteval std::size_t parse_index(std::string_view token) {
|
||||
std::size_t result = 0;
|
||||
for (char c : token) {
|
||||
result = result * 10 + (c - '0');
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
// Count tokens in pointer
|
||||
static consteval std::size_t count_tokens() {
|
||||
if (pointer_str.empty() || pointer_str == "/") return 0;
|
||||
|
||||
std::size_t count = 0;
|
||||
std::size_t pos = pointer_str[0] == '/' ? 1 : 0;
|
||||
|
||||
while (pos < pointer_str.size()) {
|
||||
++count;
|
||||
std::size_t next_slash = pointer_str.find('/', pos);
|
||||
if (next_slash == std::string_view::npos) break;
|
||||
pos = next_slash + 1;
|
||||
}
|
||||
|
||||
return count;
|
||||
}
|
||||
|
||||
// Get Nth token
|
||||
static consteval std::string_view get_token(std::size_t token_index) {
|
||||
std::size_t pos = pointer_str[0] == '/' ? 1 : 0;
|
||||
std::size_t current_token = 0;
|
||||
|
||||
while (current_token < token_index) {
|
||||
std::size_t next_slash = pointer_str.find('/', pos);
|
||||
pos = next_slash + 1;
|
||||
++current_token;
|
||||
}
|
||||
|
||||
std::size_t token_end = pointer_str.find('/', pos);
|
||||
if (token_end == std::string_view::npos) token_end = pointer_str.size();
|
||||
|
||||
return pointer_str.substr(pos, token_end - pos);
|
||||
}
|
||||
};
|
||||
|
||||
// JSON Pointer accessor
|
||||
template<typename T, constevalutil::fixed_string Pointer>
|
||||
struct pointer_accessor {
|
||||
using parser = json_pointer_parser<Pointer>;
|
||||
static constexpr std::string_view pointer_view = Pointer.view();
|
||||
static constexpr std::size_t token_count = parser::count_tokens();
|
||||
|
||||
// Validate pointer against struct definition
|
||||
static consteval bool validate_pointer() {
|
||||
if constexpr (!std::is_class_v<T>) {
|
||||
return true;
|
||||
}
|
||||
|
||||
auto current_type = ^^T;
|
||||
std::size_t pos = pointer_view[0] == '/' ? 1 : 0;
|
||||
|
||||
while (pos < pointer_view.size()) {
|
||||
// Extract token up to next /
|
||||
std::size_t token_end = pointer_view.find('/', pos);
|
||||
if (token_end == std::string_view::npos) token_end = pointer_view.size();
|
||||
|
||||
std::string_view token = pointer_view.substr(pos, token_end - pos);
|
||||
|
||||
if (parser::is_numeric(token)) {
|
||||
if (!path_accessor<T, Pointer>::is_array_like_reflected(current_type)) {
|
||||
return false;
|
||||
}
|
||||
current_type = path_accessor<T, Pointer>::get_element_type_reflected(current_type);
|
||||
} else {
|
||||
bool found = false;
|
||||
auto members = std::meta::nonstatic_data_members_of(current_type, std::meta::access_context::unchecked());
|
||||
|
||||
for (auto mem : members) {
|
||||
if (std::meta::identifier_of(mem) == token) {
|
||||
current_type = std::meta::type_of(mem);
|
||||
found = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if (!found) return false;
|
||||
}
|
||||
|
||||
pos = token_end + 1;
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
// Recursive accessor
|
||||
template<std::size_t TokenIndex>
|
||||
static inline simdjson_result<value> access_impl(simdjson_result<value> current) noexcept {
|
||||
if constexpr (TokenIndex >= token_count) {
|
||||
return current;
|
||||
} else {
|
||||
constexpr std::string_view token = parser::get_token(TokenIndex);
|
||||
|
||||
if constexpr (parser::is_numeric(token)) {
|
||||
constexpr std::size_t index = parser::parse_index(token);
|
||||
auto arr = current.get_array().value_unsafe();
|
||||
auto next_value = arr.at(index);
|
||||
return access_impl<TokenIndex + 1>(next_value);
|
||||
} else {
|
||||
auto obj = current.get_object().value_unsafe();
|
||||
auto next_value = obj.find_field_unordered(token);
|
||||
return access_impl<TokenIndex + 1>(next_value);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Access JSON value at pointer
|
||||
template<typename DocOrValue>
|
||||
static inline simdjson_result<value> access(DocOrValue& doc_or_val) noexcept {
|
||||
if constexpr (std::is_class_v<T>) {
|
||||
constexpr bool pointer_valid = validate_pointer();
|
||||
static_assert(pointer_valid, "JSON Pointer does not match struct definition");
|
||||
}
|
||||
|
||||
if (pointer_view.empty() || pointer_view == "/") {
|
||||
if constexpr (requires { doc_or_val.get_value(); }) {
|
||||
return doc_or_val.get_value();
|
||||
} else {
|
||||
return doc_or_val;
|
||||
}
|
||||
}
|
||||
|
||||
simdjson_result<value> current = doc_or_val.get_value();
|
||||
return access_impl<0>(current);
|
||||
}
|
||||
|
||||
// Extract value at pointer directly into target with type validation
|
||||
template<typename DocOrValue, typename FieldType>
|
||||
static inline error_code extract_field(DocOrValue& doc_or_val, FieldType& target) noexcept {
|
||||
static_assert(std::is_class_v<T>, "extract_field requires T to be a struct type for validation");
|
||||
|
||||
constexpr bool pointer_valid = validate_pointer();
|
||||
static_assert(pointer_valid, "JSON Pointer does not match struct definition");
|
||||
|
||||
constexpr auto final_type = get_final_type();
|
||||
static_assert(final_type == ^^FieldType, "Target type does not match the field type at the pointer");
|
||||
|
||||
simdjson_result<value> current_value = doc_or_val.get_value();
|
||||
auto json_value = access_impl<0>(current_value);
|
||||
if (json_value.error()) return json_value.error();
|
||||
|
||||
return json_value.get(target);
|
||||
}
|
||||
|
||||
private:
|
||||
// Get final type by walking pointer through struct
|
||||
template<typename U = T>
|
||||
static consteval std::enable_if_t<std::is_class_v<U>, std::meta::info> get_final_type() {
|
||||
auto current_type = ^^T;
|
||||
std::size_t pos = pointer_view[0] == '/' ? 1 : 0;
|
||||
|
||||
while (pos < pointer_view.size()) {
|
||||
std::size_t token_end = pointer_view.find('/', pos);
|
||||
if (token_end == std::string_view::npos) token_end = pointer_view.size();
|
||||
|
||||
std::string_view token = pointer_view.substr(pos, token_end - pos);
|
||||
|
||||
if (parser::is_numeric(token)) {
|
||||
current_type = path_accessor<T, "">::get_element_type_reflected(current_type);
|
||||
} else {
|
||||
auto members = std::meta::nonstatic_data_members_of(current_type, std::meta::access_context::unchecked());
|
||||
|
||||
for (auto mem : members) {
|
||||
if (std::meta::identifier_of(mem) == token) {
|
||||
current_type = std::meta::type_of(mem);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pos = token_end + 1;
|
||||
}
|
||||
|
||||
return current_type;
|
||||
}
|
||||
};
|
||||
|
||||
// Compile-time JSON Pointer accessor with validation
|
||||
template<typename T, constevalutil::fixed_string Pointer, typename DocOrValue>
|
||||
inline simdjson_result<::simdjson::SIMDJSON_IMPLEMENTATION::ondemand::value> at_pointer_compiled(DocOrValue& doc_or_val) noexcept {
|
||||
using accessor = pointer_accessor<T, Pointer>;
|
||||
return accessor::access(doc_or_val);
|
||||
}
|
||||
|
||||
// Overload without type parameter (no validation)
|
||||
template<constevalutil::fixed_string Pointer, typename DocOrValue>
|
||||
inline simdjson_result<::simdjson::SIMDJSON_IMPLEMENTATION::ondemand::value> at_pointer_compiled(DocOrValue& doc_or_val) noexcept {
|
||||
using accessor = pointer_accessor<void, Pointer>;
|
||||
return accessor::access(doc_or_val);
|
||||
}
|
||||
|
||||
} // namespace json_path
|
||||
} // namespace ondemand
|
||||
} // namespace SIMDJSON_IMPLEMENTATION
|
||||
} // namespace simdjson
|
||||
|
||||
#endif // SIMDJSON_SUPPORTS_CONCEPTS && SIMDJSON_STATIC_REFLECTION
|
||||
#endif // SIMDJSON_GENERIC_ONDEMAND_COMPILE_TIME_ACCESSORS_H
|
||||
|
||||
@@ -8,7 +8,6 @@
|
||||
// Internal headers needed for ondemand generics.
|
||||
// All includes not under simdjson/generic/ondemand must be here!
|
||||
// Otherwise, amalgamation will fail.
|
||||
#include "simdjson/concepts.h"
|
||||
#include "simdjson/dom/base.h" // for MINIMAL_DOCUMENT_CAPACITY
|
||||
#include "simdjson/implementation.h"
|
||||
#include "simdjson/padded_string.h"
|
||||
|
||||
@@ -347,7 +347,22 @@ simdjson_inline simdjson_result<value> document::at_path(std::string_view json_p
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
simdjson_inline simdjson_result<std::vector<value>> document::at_path_with_wildcard(std::string_view json_path) noexcept {
|
||||
rewind(); // Rewind the document each time at_path_with_wildcard is called
|
||||
if (json_path.empty()) {
|
||||
return INVALID_JSON_POINTER;
|
||||
}
|
||||
json_type t;
|
||||
SIMDJSON_TRY(type().get(t));
|
||||
switch (t) {
|
||||
case json_type::array:
|
||||
return (*this).get_array().at_path_with_wildcard(json_path);
|
||||
case json_type::object:
|
||||
return (*this).get_object().at_path_with_wildcard(json_path);
|
||||
default:
|
||||
return INVALID_JSON_POINTER;
|
||||
}
|
||||
}
|
||||
|
||||
#if SIMDJSON_SUPPORTS_CONCEPTS && SIMDJSON_STATIC_REFLECTION
|
||||
|
||||
@@ -672,6 +687,11 @@ simdjson_inline simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::value> simdjs
|
||||
return first.at_path(json_path);
|
||||
}
|
||||
|
||||
simdjson_inline simdjson_result<std::vector<SIMDJSON_IMPLEMENTATION::ondemand::value>> simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::document>::at_path_with_wildcard(std::string_view json_path) noexcept {
|
||||
if (error()) { return error(); }
|
||||
return first.at_path_with_wildcard(json_path);
|
||||
}
|
||||
|
||||
#if SIMDJSON_STATIC_REFLECTION
|
||||
template<constevalutil::fixed_string... FieldNames, typename T>
|
||||
requires(std::is_class_v<T> && (sizeof...(FieldNames) > 0))
|
||||
@@ -766,6 +786,7 @@ simdjson_inline simdjson_result<number> document_reference::get_number() noexcep
|
||||
simdjson_inline simdjson_result<std::string_view> document_reference::raw_json_token() noexcept { return doc->raw_json_token(); }
|
||||
simdjson_inline simdjson_result<value> document_reference::at_pointer(std::string_view json_pointer) noexcept { return doc->at_pointer(json_pointer); }
|
||||
simdjson_inline simdjson_result<value> document_reference::at_path(std::string_view json_path) noexcept { return doc->at_path(json_path); }
|
||||
simdjson_inline simdjson_result<std::vector<value>> document_reference::at_path_with_wildcard(std::string_view json_path) noexcept { return doc->at_path_with_wildcard(json_path); }
|
||||
simdjson_inline simdjson_result<std::string_view> document_reference::raw_json() noexcept { return doc->raw_json();}
|
||||
simdjson_inline document_reference::operator document&() const noexcept { return *doc; }
|
||||
#if SIMDJSON_SUPPORTS_CONCEPTS && SIMDJSON_STATIC_REFLECTION
|
||||
@@ -1022,6 +1043,12 @@ simdjson_inline simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::value> simdjs
|
||||
}
|
||||
return first.at_path(json_path);
|
||||
}
|
||||
simdjson_inline simdjson_result<std::vector<SIMDJSON_IMPLEMENTATION::ondemand::value>> simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::document_reference>::at_path_with_wildcard(std::string_view json_path) noexcept {
|
||||
if (error()) {
|
||||
return error();
|
||||
}
|
||||
return first.at_path_with_wildcard(json_path);
|
||||
}
|
||||
#if SIMDJSON_STATIC_REFLECTION
|
||||
template<constevalutil::fixed_string... FieldNames, typename T>
|
||||
requires(std::is_class_v<T> && (sizeof...(FieldNames) > 0))
|
||||
|
||||
@@ -6,6 +6,7 @@
|
||||
#include "simdjson/generic/ondemand/json_iterator.h"
|
||||
#include "simdjson/generic/ondemand/deserialize.h"
|
||||
#include "simdjson/generic/ondemand/value.h"
|
||||
#include <vector>
|
||||
#endif // SIMDJSON_CONDITIONAL_INCLUDE
|
||||
|
||||
|
||||
@@ -402,12 +403,14 @@ public:
|
||||
simdjson_inline simdjson_result<array_iterator> end() & noexcept;
|
||||
|
||||
/**
|
||||
* Look up a field by name on an object (order-sensitive).
|
||||
* Look up a field by name on an object (order-sensitive). By order-sensitive, we mean that
|
||||
* fields must be accessed in the order they appear in the JSON text (although you can
|
||||
* skip fields). See find_field_unordered() and operator[] for an order-insensitive version.
|
||||
*
|
||||
* The following code reads z, then y, then x, and thus will not retrieve x or y if fed the
|
||||
* JSON `{ "x": 1, "y": 2, "z": 3 }`:
|
||||
*
|
||||
* ```c++
|
||||
* ```cpp
|
||||
* simdjson::ondemand::parser parser;
|
||||
* auto obj = parser.parse(R"( { "x": 1, "y": 2, "z": 3 } )"_padded);
|
||||
* double z = obj.find_field("z");
|
||||
@@ -446,7 +449,8 @@ public:
|
||||
* missing case has a non-cache-friendly bump and lots of extra scanning, especially if the object
|
||||
* in question is large. The fact that the extra code is there also bumps the executable size.
|
||||
*
|
||||
* It is the default, however, because it would be highly surprising (and hard to debug) if the
|
||||
* We default operator[] on find_field_unordered() for convenience.
|
||||
* It is the default because it would be highly surprising (and hard to debug) if the
|
||||
* default behavior failed to look up a field just because it was in the wrong order--and many
|
||||
* APIs assume this. Therefore, you must be explicit if you want to treat objects as out of order.
|
||||
*
|
||||
@@ -701,7 +705,7 @@ public:
|
||||
* JSONPath queries that trivially convertible to JSON Pointer queries: key
|
||||
* names and array indices.
|
||||
*
|
||||
* https://datatracker.ietf.org/doc/html/draft-normington-jsonpath-00
|
||||
* https://www.rfc-editor.org/rfc/rfc9535 (RFC 9535)
|
||||
*
|
||||
* Key values are matched exactly, without unescaping or Unicode normalization.
|
||||
* We do a byte-by-byte comparison. E.g.
|
||||
@@ -719,6 +723,23 @@ public:
|
||||
*/
|
||||
simdjson_inline simdjson_result<value> at_path(std::string_view json_path) noexcept;
|
||||
|
||||
/**
|
||||
* Get all values matching the given JSONPath expression with wildcard support.
|
||||
*
|
||||
* Supports wildcard patterns like "$.array[*]" or "$.object.*" to match multiple elements.
|
||||
*
|
||||
* This method materializes all matching values into a vector.
|
||||
* The document will be consumed after this call.
|
||||
*
|
||||
* @param json_path JSONPath expression with wildcards
|
||||
* @return Vector of values matching the wildcard pattern, or:
|
||||
* - INVALID_JSON_POINTER if the JSONPath cannot be parsed
|
||||
* - NO_SUCH_FIELD if a field does not exist
|
||||
* - INDEX_OUT_OF_BOUNDS if an array index is out of bounds
|
||||
* - INCORRECT_TYPE if path traversal encounters wrong type
|
||||
*/
|
||||
simdjson_inline simdjson_result<std::vector<value>> at_path_with_wildcard(std::string_view json_path) noexcept;
|
||||
|
||||
/**
|
||||
* Consumes the document and returns a string_view instance corresponding to the
|
||||
* document as represented in JSON. It points inside the original byte array containing
|
||||
@@ -734,7 +755,7 @@ public:
|
||||
* potentially improving performance by skipping unwanted fields.
|
||||
*
|
||||
* Example:
|
||||
* ```c++
|
||||
* ```cpp
|
||||
* struct Car {
|
||||
* std::string make;
|
||||
* std::string model;
|
||||
@@ -936,6 +957,7 @@ public:
|
||||
simdjson_inline simdjson_result<std::string_view> raw_json_token() noexcept;
|
||||
simdjson_inline simdjson_result<value> at_pointer(std::string_view json_pointer) noexcept;
|
||||
simdjson_inline simdjson_result<value> at_path(std::string_view json_path) noexcept;
|
||||
simdjson_inline simdjson_result<std::vector<value>> at_path_with_wildcard(std::string_view json_path) noexcept;
|
||||
|
||||
private:
|
||||
document *doc{nullptr};
|
||||
@@ -1019,6 +1041,7 @@ public:
|
||||
|
||||
simdjson_inline simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::value> at_pointer(std::string_view json_pointer) noexcept;
|
||||
simdjson_inline simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::value> at_path(std::string_view json_path) noexcept;
|
||||
simdjson_inline simdjson_result<std::vector<SIMDJSON_IMPLEMENTATION::ondemand::value>> at_path_with_wildcard(std::string_view json_path) noexcept;
|
||||
#if SIMDJSON_STATIC_REFLECTION
|
||||
template<constevalutil::fixed_string... FieldNames, typename T>
|
||||
requires(std::is_class_v<T> && (sizeof...(FieldNames) > 0))
|
||||
@@ -1101,6 +1124,7 @@ public:
|
||||
|
||||
simdjson_inline simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::value> at_pointer(std::string_view json_pointer) noexcept;
|
||||
simdjson_inline simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::value> at_path(std::string_view json_path) noexcept;
|
||||
simdjson_inline simdjson_result<std::vector<SIMDJSON_IMPLEMENTATION::ondemand::value>> at_path_with_wildcard(std::string_view json_path) noexcept;
|
||||
#if SIMDJSON_STATIC_REFLECTION
|
||||
template<constevalutil::fixed_string... FieldNames, typename T>
|
||||
requires(std::is_class_v<T> && (sizeof...(FieldNames) > 0))
|
||||
|
||||
@@ -81,7 +81,7 @@ public:
|
||||
/**
|
||||
* Construct an uninitialized document_stream.
|
||||
*
|
||||
* ```c++
|
||||
* ```cpp
|
||||
* document_stream docs;
|
||||
* auto error = parser.iterate_many(json).get(docs);
|
||||
* ```
|
||||
|
||||
@@ -215,11 +215,10 @@ simdjson_inline void json_iterator::assert_more_tokens(uint32_t required_tokens)
|
||||
}
|
||||
|
||||
simdjson_inline void json_iterator::assert_valid_position(token_position position) const noexcept {
|
||||
(void)position; // Suppress unused parameter warning
|
||||
#ifndef SIMDJSON_CLANG_VISUAL_STUDIO
|
||||
SIMDJSON_ASSUME( position >= &parser->implementation->structural_indexes[0] );
|
||||
SIMDJSON_ASSUME( position < &parser->implementation->structural_indexes[parser->implementation->n_structural_indexes] );
|
||||
#else
|
||||
(void)position; // Suppress unused parameter warning
|
||||
#endif
|
||||
}
|
||||
|
||||
@@ -358,17 +357,25 @@ simdjson_inline token_position json_iterator::position() const noexcept {
|
||||
simdjson_inline simdjson_result<std::string_view> json_iterator::unescape(raw_json_string in, bool allow_replacement) noexcept {
|
||||
#if SIMDJSON_DEVELOPMENT_CHECKS
|
||||
auto result = parser->unescape(in, _string_buf_loc, allow_replacement);
|
||||
#if !defined(SIMDJSON_VISUAL_STUDIO) && !defined(SIMDJSON_CLANG_VISUAL_STUDIO)
|
||||
// Under Visual Studio, the next SIMDJSON_ASSUME fails with: the argument
|
||||
// has side effects that will be discarded.
|
||||
SIMDJSON_ASSUME(!parser->string_buffer_overflow(_string_buf_loc));
|
||||
#endif // !defined(SIMDJSON_VISUAL_STUDIO) && !defined(SIMDJSON_CLANG_VISUAL_STUDIO)
|
||||
return result;
|
||||
#else
|
||||
return parser->unescape(in, _string_buf_loc, allow_replacement);
|
||||
return parser->unescape_maybe(in, _string_buf_loc, allow_replacement);
|
||||
#endif
|
||||
}
|
||||
|
||||
simdjson_inline simdjson_result<std::string_view> json_iterator::unescape_wobbly(raw_json_string in) noexcept {
|
||||
#if SIMDJSON_DEVELOPMENT_CHECKS
|
||||
auto result = parser->unescape_wobbly(in, _string_buf_loc);
|
||||
#if !defined(SIMDJSON_VISUAL_STUDIO) && !defined(SIMDJSON_CLANG_VISUAL_STUDIO)
|
||||
// Under Visual Studio, the next SIMDJSON_ASSUME fails with: the argument
|
||||
// has side effects that will be discarded.
|
||||
SIMDJSON_ASSUME(!parser->string_buffer_overflow(_string_buf_loc));
|
||||
#endif // !defined(SIMDJSON_VISUAL_STUDIO) && !defined(SIMDJSON_CLANG_VISUAL_STUDIO)
|
||||
return result;
|
||||
#else
|
||||
return parser->unescape_wobbly(in, _string_buf_loc);
|
||||
|
||||
@@ -9,6 +9,7 @@
|
||||
#include "simdjson/generic/ondemand/raw_json_string.h"
|
||||
#include "simdjson/generic/ondemand/json_iterator.h"
|
||||
#include "simdjson/generic/ondemand/value-inl.h"
|
||||
#include "simdjson/jsonpathutil.h"
|
||||
#if SIMDJSON_STATIC_REFLECTION
|
||||
#include "simdjson/generic/ondemand/json_string_builder.h" // for constevalutil::fixed_string
|
||||
#include <meta>
|
||||
@@ -176,6 +177,50 @@ inline simdjson_result<value> object::at_path(std::string_view json_path) noexce
|
||||
return at_pointer(json_pointer);
|
||||
}
|
||||
|
||||
inline simdjson_result<std::vector<value>> object::at_path_with_wildcard(std::string_view json_path) noexcept {
|
||||
std::vector<value> result;
|
||||
|
||||
auto result_pair = get_next_key_and_json_path(json_path);
|
||||
std::string_view key = result_pair.first;
|
||||
std::string_view remaining_path = result_pair.second;
|
||||
// Handle when its the case for wildcard
|
||||
if (key == "*") {
|
||||
// Loop through each field in the object
|
||||
for (auto field : *this) {
|
||||
value val;
|
||||
SIMDJSON_TRY(field.value().get(val));
|
||||
|
||||
if (remaining_path.empty()) {
|
||||
result.push_back(std::move(val));
|
||||
} else {
|
||||
auto nested_result = val.at_path_with_wildcard(remaining_path);
|
||||
|
||||
if (nested_result.error()) {
|
||||
return nested_result.error();
|
||||
}
|
||||
// Extract and append all nested matches to our result
|
||||
std::vector<value> nested_vec;
|
||||
SIMDJSON_TRY(std::move(nested_result).get(nested_vec));
|
||||
|
||||
result.insert(result.end(),
|
||||
std::make_move_iterator(nested_vec.begin()),
|
||||
std::make_move_iterator(nested_vec.end()));
|
||||
}
|
||||
}
|
||||
return result;
|
||||
} else {
|
||||
value val;
|
||||
SIMDJSON_TRY(find_field(key).get(val));
|
||||
|
||||
if (remaining_path.empty()) {
|
||||
result.push_back(std::move(val));
|
||||
return result;
|
||||
} else {
|
||||
return val.at_path_with_wildcard(remaining_path);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
simdjson_inline simdjson_result<size_t> object::count_fields() & noexcept {
|
||||
size_t count{0};
|
||||
// Important: we do not consume any of the values.
|
||||
@@ -302,6 +347,11 @@ simdjson_inline simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::value> simdjs
|
||||
return first.at_path(json_path);
|
||||
}
|
||||
|
||||
simdjson_inline simdjson_result<std::vector<SIMDJSON_IMPLEMENTATION::ondemand::value>> simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::object>::at_path_with_wildcard(std::string_view json_path) noexcept {
|
||||
if (error()) { return error(); }
|
||||
return first.at_path_with_wildcard(json_path);
|
||||
}
|
||||
|
||||
inline simdjson_result<bool> simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::object>::reset() noexcept {
|
||||
if (error()) { return error(); }
|
||||
return first.reset();
|
||||
|
||||
@@ -5,6 +5,7 @@
|
||||
#include "simdjson/generic/ondemand/base.h"
|
||||
#include "simdjson/generic/implementation_simdjson_result_base.h"
|
||||
#include "simdjson/generic/ondemand/value_iterator.h"
|
||||
#include <vector>
|
||||
#if SIMDJSON_STATIC_REFLECTION && SIMDJSON_SUPPORTS_CONCEPTS
|
||||
#include "simdjson/generic/ondemand/json_string_builder.h" // for constevalutil::fixed_string
|
||||
#endif
|
||||
@@ -26,15 +27,24 @@ public:
|
||||
*/
|
||||
simdjson_inline object() noexcept = default;
|
||||
|
||||
/**
|
||||
* Get an iterator to the start of the object. We recommend using a range-based for loop.
|
||||
*
|
||||
* Using the iterator directly is also possible but error-prone and discouraged. In particular,
|
||||
* you must dereference the iterator exactly once per iteration (before calling '++').
|
||||
* Doing otherwise is unsafe and may lead to errors. You are responsible for ensuring
|
||||
*/
|
||||
simdjson_inline simdjson_result<object_iterator> begin() noexcept;
|
||||
simdjson_inline simdjson_result<object_iterator> end() noexcept;
|
||||
/**
|
||||
* Look up a field by name on an object (order-sensitive).
|
||||
* Look up a field by name on an object (order-sensitive). By order-sensitive, we mean that
|
||||
* fields must be accessed in the order they appear in the JSON text (although you can
|
||||
* skip fields). See find_field_unordered() and operator[] for an order-insensitive version.
|
||||
*
|
||||
* The following code reads z, then y, then x, and thus will not retrieve x or y if fed the
|
||||
* JSON `{ "x": 1, "y": 2, "z": 3 }`:
|
||||
*
|
||||
* ```c++
|
||||
* ```cpp
|
||||
* simdjson::ondemand::parser parser;
|
||||
* auto obj = parser.parse(R"( { "x": 1, "y": 2, "z": 3 } )"_padded);
|
||||
* double z = obj.find_field("z");
|
||||
@@ -77,7 +87,8 @@ public:
|
||||
* missing case has a non-cache-friendly bump and lots of extra scanning, especially if the object
|
||||
* in question is large. The fact that the extra code is there also bumps the executable size.
|
||||
*
|
||||
* It is the default, however, because it would be highly surprising (and hard to debug) if the
|
||||
* We default operator[] on find_field_unordered() for convenience.
|
||||
* It is the default because it would be highly surprising (and hard to debug) if the
|
||||
* default behavior failed to look up a field just because it was in the wrong order--and many
|
||||
* APIs assume this. Therefore, you must be explicit if you want to treat objects as out of order.
|
||||
*
|
||||
@@ -160,6 +171,15 @@ public:
|
||||
*/
|
||||
inline simdjson_result<value> at_path(std::string_view json_path) noexcept;
|
||||
|
||||
/**
|
||||
* Get all values matching the given JSONPath expression with wildcard support.
|
||||
* Supports wildcard patterns like ".*" to match all object fields.
|
||||
*
|
||||
* @param json_path JSONPath expression with wildcards
|
||||
* @return Vector of values matching the wildcard pattern
|
||||
*/
|
||||
inline simdjson_result<std::vector<value>> at_path_with_wildcard(std::string_view json_path) noexcept;
|
||||
|
||||
/**
|
||||
* Reset the iterator so that we are pointing back at the
|
||||
* beginning of the object. You should still consume values only once even if you
|
||||
@@ -243,7 +263,7 @@ public:
|
||||
* potentially improving performance by skipping unwanted fields.
|
||||
*
|
||||
* Example:
|
||||
* ```c++
|
||||
* ```cpp
|
||||
* struct Car {
|
||||
* std::string make;
|
||||
* std::string model;
|
||||
@@ -308,6 +328,7 @@ public:
|
||||
simdjson_inline simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::value> operator[](std::string_view key) && noexcept;
|
||||
simdjson_inline simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::value> at_pointer(std::string_view json_pointer) noexcept;
|
||||
simdjson_inline simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::value> at_path(std::string_view json_path) noexcept;
|
||||
simdjson_inline simdjson_result<std::vector<SIMDJSON_IMPLEMENTATION::ondemand::value>> at_path_with_wildcard(std::string_view json_path) noexcept;
|
||||
inline simdjson_result<bool> reset() noexcept;
|
||||
inline simdjson_result<bool> is_empty() noexcept;
|
||||
inline simdjson_result<size_t> count_fields() & noexcept;
|
||||
|
||||
@@ -21,6 +21,11 @@ simdjson_inline object_iterator::object_iterator(const value_iterator &_iter) no
|
||||
{}
|
||||
|
||||
simdjson_inline simdjson_result<field> object_iterator::operator*() noexcept {
|
||||
#if SIMDJSON_DEVELOPMENT_CHECKS
|
||||
// We must call * once per iteration.
|
||||
SIMDJSON_ASSUME(!has_been_referenced);
|
||||
has_been_referenced = true;
|
||||
#endif
|
||||
error_code error = iter.error();
|
||||
if (error) { iter.abandon(); return error; }
|
||||
auto result = field::start(iter);
|
||||
@@ -39,6 +44,11 @@ simdjson_inline bool object_iterator::operator!=(const object_iterator &) const
|
||||
SIMDJSON_PUSH_DISABLE_WARNINGS
|
||||
SIMDJSON_DISABLE_STRICT_OVERFLOW_WARNING
|
||||
simdjson_inline object_iterator &object_iterator::operator++() noexcept {
|
||||
#if SIMDJSON_DEVELOPMENT_CHECKS
|
||||
// Before calling ++, we must have called *.
|
||||
SIMDJSON_ASSUME(has_been_referenced);
|
||||
has_been_referenced = false;
|
||||
#endif
|
||||
// TODO this is a safety rail ... users should exit loops as soon as they receive an error.
|
||||
// Nonetheless, let's see if performance is OK with this if statement--the compiler may give it to us for free.
|
||||
if (!iter.is_open()) { return *this; } // Iterator will be released if there is an error
|
||||
|
||||
@@ -32,9 +32,14 @@ public:
|
||||
// Assumes it's being compared with the end. true if depth >= iter->depth.
|
||||
simdjson_inline bool operator!=(const object_iterator &) const noexcept;
|
||||
// Checks for ']' and ','
|
||||
// YOU MUST NOT CALL THIS IF operator* YIELDED AN ERROR.
|
||||
// YOU MUST NOT CALL THIS WITHOUT A CORRESPONDING operator* CALL.
|
||||
simdjson_inline object_iterator &operator++() noexcept;
|
||||
|
||||
private:
|
||||
#if SIMDJSON_DEVELOPMENT_CHECKS
|
||||
bool has_been_referenced{false};
|
||||
#endif
|
||||
/**
|
||||
* The underlying JSON iterator.
|
||||
*
|
||||
|
||||