mirror of
https://github.com/simdjson/simdjson
synced 2026-06-08 17:27:07 +00:00
Compare commits
75 Commits
v3.9.1
...
movemsgpack
| Author | SHA1 | Date | |
|---|---|---|---|
| 969525cd7d | |||
| 36f5dbcb75 | |||
| 49901fb254 | |||
| c066b5421b | |||
| ec7550a70e | |||
| 3ef3078e51 | |||
| fd06782c97 | |||
| 9303efbd0c | |||
| 6c979f15cc | |||
| 09ccabbe6c | |||
| e00cc8c6dc | |||
| c10b32d463 | |||
| 025a44348a | |||
| 70a68da941 | |||
| e341c8b438 | |||
| 0ac0a80e28 | |||
| 6b9117c029 | |||
| dd92151971 | |||
| 0679c247f4 | |||
| 615218a3ad | |||
| 4c1b0a41d8 | |||
| ef563a4b09 | |||
| 7a9ff93388 | |||
| fc61d7c7ba | |||
| d506af0a79 | |||
| ccf8694510 | |||
| 9b67497ed0 | |||
| 0336684df7 | |||
| c19320dd6e | |||
| 412a5680e8 | |||
| a05a56856d | |||
| 58173a6a1f | |||
| 1721032cfd | |||
| b73877f95e | |||
| 09723897e9 | |||
| 49e231b634 | |||
| acdbbab916 | |||
| 5090247c34 | |||
| 4180e05730 | |||
| feea2bce2c | |||
| 692f43cd84 | |||
| 3240d55bcc | |||
| 66fd28fc00 | |||
| 5f638951c6 | |||
| 3e94eea939 | |||
| 0e8311f812 | |||
| 3620e9d151 | |||
| d017cd7ca4 | |||
| d9d1ff5856 | |||
| eb8f2bce14 | |||
| 2a4ff73468 | |||
| 77fc2b8447 | |||
| ba8b66a633 | |||
| 66eec5feaf | |||
| 3964f3e5d2 | |||
| ee8515122d | |||
| 5d35e7ca1f | |||
| deefc88b9c | |||
| c80dda7c58 | |||
| d2954ef68b | |||
| ac719827ff | |||
| 6ea77392a7 | |||
| e2f879751c | |||
| bf7834179c | |||
| 69ab8848bf | |||
| 7288323e36 | |||
| 3f381cf3ee | |||
| 52406402ed | |||
| 799d8e3fc9 | |||
| 6c647d9f3b | |||
| 8519e24f12 | |||
| fffb62743c | |||
| c888075f8d | |||
| 9b8aac5f22 | |||
| 40b414d184 |
@@ -25,6 +25,7 @@ CompileFlags:
|
|||||||
Diagnostics:
|
Diagnostics:
|
||||||
Suppress:
|
Suppress:
|
||||||
- pp_including_mainfile_in_preamble
|
- pp_including_mainfile_in_preamble
|
||||||
|
- unused-includes
|
||||||
---
|
---
|
||||||
# Amalgamated files that require or partly define an implementation
|
# Amalgamated files that require or partly define an implementation
|
||||||
If:
|
If:
|
||||||
|
|||||||
@@ -0,0 +1 @@
|
|||||||
|
blank_issues_enabled: false
|
||||||
@@ -18,7 +18,7 @@ We do not make changes to simdjson without clearly identifiable benefits, which
|
|||||||
|
|
||||||
Is your issue:
|
Is your issue:
|
||||||
|
|
||||||
1. A bug report? If so, please point at a reproducible test. Indicate whether you are willing or able to provide a bug fix as a pull request.
|
1. A bug report? If so, please point at a reproducible test. Indicate whether you are willing or able to provide a bug fix as a pull request. As a matter of policy, we do not consider a compiler warning to be a bug.
|
||||||
|
|
||||||
2. A build issue? If so, provide all possible details regarding your system configuration. If we cannot reproduce your issue, we cannot fix it.
|
2. A build issue? If so, provide all possible details regarding your system configuration. If we cannot reproduce your issue, we cannot fix it.
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,29 @@
|
|||||||
|
name: Ubuntu ppc64le (GCC 11)
|
||||||
|
|
||||||
|
on:
|
||||||
|
push:
|
||||||
|
branches:
|
||||||
|
- master
|
||||||
|
pull_request:
|
||||||
|
branches:
|
||||||
|
- master
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
build:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@v4
|
||||||
|
- uses: uraimo/run-on-arch-action@v2
|
||||||
|
name: Test
|
||||||
|
id: runcmd
|
||||||
|
with:
|
||||||
|
arch: aarch64
|
||||||
|
distro: ubuntu_latest
|
||||||
|
githubToken: ${{ github.token }}
|
||||||
|
install: |
|
||||||
|
apt-get update -q -y
|
||||||
|
apt-get install -y cmake make g++
|
||||||
|
run: |
|
||||||
|
cmake -DSIMDJSON_SANITIZE_UNDEFINED=ON -DSIMDJSON_DEVELOPER_MODE=ON -DSIMDJSON_COMPETITION=OFF -B build
|
||||||
|
cmake --build build -j=2
|
||||||
|
ctest --output-on-failure --test-dir build
|
||||||
@@ -17,7 +17,7 @@ jobs:
|
|||||||
fuzz-seconds: 600
|
fuzz-seconds: 600
|
||||||
dry-run: false
|
dry-run: false
|
||||||
- name: Upload Crash
|
- name: Upload Crash
|
||||||
uses: actions/upload-artifact@v1
|
uses: actions/upload-artifact@v4
|
||||||
if: failure() && steps.build.outcome == 'success'
|
if: failure() && steps.build.outcome == 'success'
|
||||||
with:
|
with:
|
||||||
name: artifacts
|
name: artifacts
|
||||||
|
|||||||
@@ -24,7 +24,7 @@ jobs:
|
|||||||
echo "no trailing whitespace found, good!"
|
echo "no trailing whitespace found, good!"
|
||||||
fi
|
fi
|
||||||
- name: Archive whitespace patch
|
- name: Archive whitespace patch
|
||||||
uses: actions/upload-artifact@v2
|
uses: actions/upload-artifact@v4
|
||||||
if: always()
|
if: always()
|
||||||
with:
|
with:
|
||||||
name: whitespace-patch
|
name: whitespace-patch
|
||||||
|
|||||||
@@ -20,6 +20,9 @@ jobs:
|
|||||||
- msystem: "MINGW64"
|
- msystem: "MINGW64"
|
||||||
install: mingw-w64-x86_64-libxml2 mingw-w64-x86_64-cmake mingw-w64-x86_64-ninja mingw-w64-x86_64-clang
|
install: mingw-w64-x86_64-libxml2 mingw-w64-x86_64-cmake mingw-w64-x86_64-ninja mingw-w64-x86_64-clang
|
||||||
type: Debug
|
type: Debug
|
||||||
|
- msystem: "MINGW64"
|
||||||
|
install: mingw-w64-x86_64-libxml2 mingw-w64-x86_64-cmake mingw-w64-x86_64-ninja mingw-w64-x86_64-clang
|
||||||
|
type: RelWithDebInfo
|
||||||
env:
|
env:
|
||||||
CMAKE_GENERATOR: Ninja
|
CMAKE_GENERATOR: Ninja
|
||||||
|
|
||||||
|
|||||||
@@ -22,6 +22,9 @@ jobs:
|
|||||||
- msystem: "MINGW64"
|
- msystem: "MINGW64"
|
||||||
install: mingw-w64-x86_64-cmake mingw-w64-x86_64-ninja mingw-w64-x86_64-gcc
|
install: mingw-w64-x86_64-cmake mingw-w64-x86_64-ninja mingw-w64-x86_64-gcc
|
||||||
type: Debug
|
type: Debug
|
||||||
|
- msystem: "MINGW64"
|
||||||
|
install: mingw-w64-x86_64-cmake mingw-w64-x86_64-ninja mingw-w64-x86_64-gcc
|
||||||
|
type: RelWithDebInfo
|
||||||
env:
|
env:
|
||||||
CMAKE_GENERATOR: Ninja
|
CMAKE_GENERATOR: Ninja
|
||||||
|
|
||||||
|
|||||||
@@ -1,71 +0,0 @@
|
|||||||
name: short fuzz on the power arch
|
|
||||||
|
|
||||||
on:
|
|
||||||
push:
|
|
||||||
branches: [ master ]
|
|
||||||
pull_request:
|
|
||||||
branches: [ master ]
|
|
||||||
|
|
||||||
jobs:
|
|
||||||
armv7_job:
|
|
||||||
if: >-
|
|
||||||
! contains(toJSON(github.event.commits.*.message), '[skip ci]') &&
|
|
||||||
! contains(toJSON(github.event.commits.*.message), '[skip github]')
|
|
||||||
# The host should always be Linux
|
|
||||||
runs-on: ubuntu-20.04
|
|
||||||
name: Build on ubuntu-20.04 ppc64le
|
|
||||||
steps:
|
|
||||||
- uses: actions/checkout@v4
|
|
||||||
- uses: uraimo/run-on-arch-action@v2.0.5
|
|
||||||
name: Run commands
|
|
||||||
id: runcmd
|
|
||||||
env:
|
|
||||||
DEBIAN_FRONTEND: noninteractive
|
|
||||||
with:
|
|
||||||
arch: ppc64le
|
|
||||||
distro: buster
|
|
||||||
|
|
||||||
# Not required, but speeds up builds by storing container images in
|
|
||||||
# a GitHub package registry.
|
|
||||||
githubToken: ${{ github.token }}
|
|
||||||
|
|
||||||
run: |
|
|
||||||
export CLANGSUFFIX="-7"
|
|
||||||
apt-get -qq update
|
|
||||||
apt-get install -q -y clang-7 libfuzzer-7-dev git wget zip ninja-build gnupg software-properties-common
|
|
||||||
wget -q -O - "https://raw.githubusercontent.com/simdjson/debian-ppa/master/key.gpg" | apt-key add -
|
|
||||||
apt-add-repository "deb https://raw.githubusercontent.com/simdjson/debian-ppa/master simdjson main"
|
|
||||||
apt-get -qq update
|
|
||||||
apt-get purge cmake cmake-data
|
|
||||||
apt-get -t simdjson -y install cmake
|
|
||||||
mkdir -p build ; cd build
|
|
||||||
cmake .. -GNinja \
|
|
||||||
-DCMAKE_CXX_COMPILER=clang++$CLANGSUFFIX \
|
|
||||||
-DCMAKE_C_COMPILER=clang$CLANGSUFFIX \
|
|
||||||
-DBUILD_SHARED_LIBS=OFF \
|
|
||||||
-DSIMDJSON_DEVELOPER_MODE=ON \
|
|
||||||
-DSIMDJSON_ENABLE_FUZZING=On \
|
|
||||||
-DSIMDJSON_COMPETITION=OFF \
|
|
||||||
-DSIMDJSON_GOOGLE_BENCHMARKS=OFF \
|
|
||||||
-DSIMDJSON_DISABLE_DEPRECATED_API=On \
|
|
||||||
-DSIMDJSON_FUZZ_LDFLAGS=-lFuzzer \
|
|
||||||
-DCMAKE_CXX_FLAGS="-fsanitize=fuzzer-no-link -DFUZZING_BUILD_MODE_UNSAFE_FOR_PRODUCTION=" \
|
|
||||||
-DCMAKE_C_FLAGS="-fsanitize=fuzzer-no-link" \
|
|
||||||
-DCMAKE_BUILD_TYPE=Release \
|
|
||||||
-DSIMDJSON_FUZZ_LINKMAIN=Off
|
|
||||||
cd ..
|
|
||||||
builddir=build
|
|
||||||
cmake --build $builddir
|
|
||||||
wget -O corpus.tar.gz https://readonly:readonly@www.pauldreik.se/fuzzdata/index.php?project=simdjson
|
|
||||||
tar xf corpus.tar.gz
|
|
||||||
fuzzernames=$(cmake --build $builddir --target print_all_fuzzernames |tail -n1)
|
|
||||||
for fuzzer in $fuzzernames ; do
|
|
||||||
exe=$builddir/fuzz/$fuzzer
|
|
||||||
shortname=$(echo $fuzzer |cut -f2- -d_)
|
|
||||||
echo found fuzzer $shortname with executable $exe
|
|
||||||
mkdir -p out/$shortname
|
|
||||||
others=$(find out -type d -not -name $shortname -not -name out -not -name cmin)
|
|
||||||
$exe -max_total_time=20 -max_len=4000 out/$shortname $others
|
|
||||||
echo "*************************************************************************"
|
|
||||||
done
|
|
||||||
echo "all is good, no errors found in any of these fuzzers: $fuzzernames"
|
|
||||||
@@ -24,6 +24,6 @@ jobs:
|
|||||||
apt-get update -q -y
|
apt-get update -q -y
|
||||||
apt-get install -y cmake make g++
|
apt-get install -y cmake make g++
|
||||||
run: |
|
run: |
|
||||||
cmake -DCMAKE_BUILD_TYPE=Release -B build
|
cmake -DCMAKE_BUILD_TYPE=Release -DSIMDJSON_DEVELOPER_MODE=ON -DSIMDJSON_COMPETITION=OFF -B build
|
||||||
cmake --build build -j=2
|
cmake --build build -j=2
|
||||||
ctest --output-on-failure --test-dir build
|
ctest --output-on-failure --test-dir build
|
||||||
|
|||||||
@@ -24,6 +24,6 @@ jobs:
|
|||||||
apt-get update -q -y
|
apt-get update -q -y
|
||||||
apt-get install -y cmake make g++
|
apt-get install -y cmake make g++
|
||||||
run: |
|
run: |
|
||||||
cmake -DCMAKE_BUILD_TYPE=Release -B build
|
cmake -DCMAKE_BUILD_TYPE=Release -DSIMDJSON_DEVELOPER_MODE=ON -DSIMDJSON_COMPETITION=OFF -B build
|
||||||
cmake --build build -j=2
|
cmake --build build -j=2
|
||||||
ctest --output-on-failure --test-dir build
|
ctest --output-on-failure --test-dir build
|
||||||
|
|||||||
@@ -24,6 +24,6 @@ jobs:
|
|||||||
apt-get update -q -y
|
apt-get update -q -y
|
||||||
apt-get install -y cmake make g++
|
apt-get install -y cmake make g++
|
||||||
run: |
|
run: |
|
||||||
cmake -DCMAKE_BUILD_TYPE=Release -B build
|
cmake -DCMAKE_BUILD_TYPE=Release -DSIMDJSON_DEVELOPER_MODE=ON -DSIMDJSON_COMPETITION=OFF -B build
|
||||||
cmake --build build -j=2
|
cmake --build build -j=2
|
||||||
ctest --output-on-failure --test-dir build
|
ctest --output-on-failure --test-dir build
|
||||||
|
|||||||
@@ -1,23 +0,0 @@
|
|||||||
name: Ubuntu 22.04 CI (GCC 13)
|
|
||||||
|
|
||||||
on: [push, pull_request]
|
|
||||||
|
|
||||||
jobs:
|
|
||||||
ubuntu-build:
|
|
||||||
if: >-
|
|
||||||
! contains(toJSON(github.event.commits.*.message), '[skip ci]') &&
|
|
||||||
! contains(toJSON(github.event.commits.*.message), '[skip github]')
|
|
||||||
runs-on: ubuntu-22.04
|
|
||||||
steps:
|
|
||||||
- uses: actions/checkout@v4
|
|
||||||
- uses: actions/cache@v4
|
|
||||||
with:
|
|
||||||
path: dependencies/.cache
|
|
||||||
key: ${{ hashFiles('dependencies/CMakeLists.txt') }}
|
|
||||||
- name: Use cmake
|
|
||||||
run: |
|
|
||||||
mkdir build &&
|
|
||||||
cd build &&
|
|
||||||
CXX=g++-13 cmake -DSIMDJSON_DEVELOPER_MODE=ON .. &&
|
|
||||||
cmake --build . &&
|
|
||||||
ctest --output-on-failure -LE explicitonly -j
|
|
||||||
@@ -0,0 +1,25 @@
|
|||||||
|
name: Ubuntu 24.04 CI
|
||||||
|
|
||||||
|
on: [push, pull_request]
|
||||||
|
jobs:
|
||||||
|
ubuntu-build:
|
||||||
|
if: >-
|
||||||
|
! contains(toJSON(github.event.commits.*.message), '[skip ci]') &&
|
||||||
|
! contains(toJSON(github.event.commits.*.message), '[skip github]')
|
||||||
|
runs-on: ubuntu-24.04
|
||||||
|
strategy:
|
||||||
|
matrix:
|
||||||
|
shared: [ON, OFF]
|
||||||
|
cxx: [g++-13, clang++-16]
|
||||||
|
sanitizer: [ON, OFF]
|
||||||
|
build_type: [RelWithDebInfo, Debug, Release]
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@a5ac7e51b41094c92402da3b24376905380afc29 # v4.1.6
|
||||||
|
- name: Prepare
|
||||||
|
run: cmake -DCMAKE_BUILD_TYPE=${{matrix.build_type}} -DSIMDJSON_DEVELOPER_MODE=ON -DSIMDJSON_SANITIZE=${{matrix.sanitizer}} -DBUILD_SHARED_LIBS=${{matrix.shared}} -B build
|
||||||
|
env:
|
||||||
|
CXX: ${{matrix.cxx}}
|
||||||
|
- name: Build
|
||||||
|
run: cmake --build build -j=2
|
||||||
|
- name: Test
|
||||||
|
run: ctest --output-on-failure --test-dir build
|
||||||
@@ -12,10 +12,11 @@ jobs:
|
|||||||
include:
|
include:
|
||||||
- {arch: ARM}
|
- {arch: ARM}
|
||||||
- {arch: ARM64}
|
- {arch: ARM64}
|
||||||
|
- {arch: ARM64EC}
|
||||||
steps:
|
steps:
|
||||||
- name: checkout
|
- name: checkout
|
||||||
uses: actions/checkout@v4
|
uses: actions/checkout@v4
|
||||||
- name: Use cmake
|
- name: Use cmake
|
||||||
run: |
|
run: |
|
||||||
cmake -A ${{ matrix.arch }} -DCMAKE_CROSSCOMPILING=1 -DSIMDJSON_DEVELOPER_MODE=ON -D SIMDJSON_GOOGLE_BENCHMARKS=OFF -DSIMDJSON_EXCEPTIONS=OFF -B build &&
|
cmake -A ${{ matrix.arch }} -DCMAKE_SYSTEM_VERSION="10.0.22621.0" -DCMAKE_CROSSCOMPILING=1 -DSIMDJSON_DEVELOPER_MODE=ON -D SIMDJSON_GOOGLE_BENCHMARKS=OFF -DSIMDJSON_EXCEPTIONS=OFF -B build &&
|
||||||
cmake --build build --verbose
|
cmake --build build --verbose
|
||||||
@@ -13,10 +13,12 @@ jobs:
|
|||||||
fail-fast: false
|
fail-fast: false
|
||||||
matrix:
|
matrix:
|
||||||
include:
|
include:
|
||||||
- {gen: Visual Studio 17 2022, arch: Win32, shared: ON}
|
- {gen: Visual Studio 17 2022, arch: Win32, shared: ON, build_type: Release}
|
||||||
- {gen: Visual Studio 17 2022, arch: Win32, shared: OFF}
|
- {gen: Visual Studio 17 2022, arch: Win32, shared: OFF, build_type: Release}
|
||||||
- {gen: Visual Studio 17 2022, arch: x64, shared: ON}
|
- {gen: Visual Studio 17 2022, arch: x64, shared: ON, build_type: Release}
|
||||||
- {gen: Visual Studio 17 2022, arch: x64, shared: OFF}
|
- {gen: Visual Studio 17 2022, arch: x64, shared: OFF, build_type: Debug}
|
||||||
|
- {gen: Visual Studio 17 2022, arch: x64, shared: OFF, build_type: Release}
|
||||||
|
- {gen: Visual Studio 17 2022, arch: x64, shared: OFF, build_type: RelWithDebInfo}
|
||||||
steps:
|
steps:
|
||||||
- name: checkout
|
- name: checkout
|
||||||
uses: actions/checkout@v4
|
uses: actions/checkout@v4
|
||||||
@@ -24,21 +26,15 @@ jobs:
|
|||||||
run: |
|
run: |
|
||||||
cmake -G "${{matrix.gen}}" -A ${{matrix.arch}} -DSIMDJSON_DEVELOPER_MODE=ON -DSIMDJSON_COMPETITION=OFF -DBUILD_SHARED_LIBS=${{matrix.shared}} -B build
|
cmake -G "${{matrix.gen}}" -A ${{matrix.arch}} -DSIMDJSON_DEVELOPER_MODE=ON -DSIMDJSON_COMPETITION=OFF -DBUILD_SHARED_LIBS=${{matrix.shared}} -B build
|
||||||
- name: Build Debug
|
- name: Build Debug
|
||||||
run: cmake --build build --config Debug --verbose
|
run: cmake --build build --config ${{matrix.build_type}} --verbose
|
||||||
- name: Build Release
|
- name: Run tests
|
||||||
run: cmake --build build --config Release --verbose
|
|
||||||
- name: Run Release tests
|
|
||||||
run: |
|
run: |
|
||||||
cd build
|
cd build
|
||||||
ctest -C Release -LE explicitonly --output-on-failure
|
ctest -C ${{matrix.build_type}} -LE explicitonly --output-on-failure
|
||||||
- name: Run Debug tests
|
|
||||||
run: |
|
|
||||||
cd build
|
|
||||||
ctest -C Debug -LE explicitonly --output-on-failure
|
|
||||||
- name: Install
|
- name: Install
|
||||||
run: |
|
run: |
|
||||||
cmake --install build --config Release
|
cmake --install build --config ${{matrix.build_type}}
|
||||||
- name: Test Installation
|
- name: Test Installation
|
||||||
run: |
|
run: |
|
||||||
cmake -G "${{matrix.gen}}" -A ${{matrix.arch}} -B build_install_test tests/installation_tests/find
|
cmake -G "${{matrix.gen}}" -A ${{matrix.arch}} -B build_install_test tests/installation_tests/find
|
||||||
cmake --build build_install_test --config Release
|
cmake --build build_install_test --config ${{matrix.build_type}}
|
||||||
@@ -13,29 +13,25 @@ jobs:
|
|||||||
fail-fast: false
|
fail-fast: false
|
||||||
matrix:
|
matrix:
|
||||||
include:
|
include:
|
||||||
- {gen: Visual Studio 17 2022, arch: x64}
|
- {gen: Visual Studio 17 2022, arch: x64, build_type: Debug}
|
||||||
|
- {gen: Visual Studio 17 2022, arch: x64, build_type: Release}
|
||||||
|
- {gen: Visual Studio 17 2022, arch: x64, build_type: RelWithDebInfo}
|
||||||
steps:
|
steps:
|
||||||
- name: checkout
|
- name: checkout
|
||||||
uses: actions/checkout@v4
|
uses: actions/checkout@v4
|
||||||
- name: Configure
|
- name: Configure
|
||||||
run: |
|
run: |
|
||||||
cmake -G "${{matrix.gen}}" -A ${{matrix.arch}} -T ClangCL -DSIMDJSON_DEVELOPER_MODE=ON -DSIMDJSON_COMPETITION=OFF -B build
|
cmake -G "${{matrix.gen}}" -A ${{matrix.arch}} -T ClangCL -DSIMDJSON_DEVELOPER_MODE=ON -DSIMDJSON_COMPETITION=OFF -B build
|
||||||
- name: Build Debug
|
- name: Build
|
||||||
run: cmake --build build --config Debug --verbose
|
run: cmake --build build --config ${{matrix.build_type}} --verbose
|
||||||
- name: Build Release
|
- name: Run tests
|
||||||
run: cmake --build build --config Release --verbose
|
|
||||||
- name: Run Release tests
|
|
||||||
run: |
|
run: |
|
||||||
cd build
|
cd build
|
||||||
ctest -C Release -LE explicitonly --output-on-failure
|
ctest -C ${{matrix.build_type}} -LE explicitonly --output-on-failure
|
||||||
- name: Run Debug tests
|
|
||||||
run: |
|
|
||||||
cd build
|
|
||||||
ctest -C Debug -LE explicitonly --output-on-failure
|
|
||||||
- name: Install
|
- name: Install
|
||||||
run: |
|
run: |
|
||||||
cmake --install build --config Release
|
cmake --install build --config ${{matrix.build_type}}
|
||||||
- name: Test Installation
|
- name: Test Installation
|
||||||
run: |
|
run: |
|
||||||
cmake -G "${{matrix.gen}}" -A ${{matrix.arch}} -B build_install_test tests/installation_tests/find
|
cmake -G "${{matrix.gen}}" -A ${{matrix.arch}} -B build_install_test tests/installation_tests/find
|
||||||
cmake --build build_install_test --config Release
|
cmake --build build_install_test --config ${{matrix.build_type}}
|
||||||
Vendored
+5
@@ -110,5 +110,10 @@
|
|||||||
"semaphore": "cpp",
|
"semaphore": "cpp",
|
||||||
"stop_token": "cpp",
|
"stop_token": "cpp",
|
||||||
"cfenv": "cpp"
|
"cfenv": "cpp"
|
||||||
|
},
|
||||||
|
"cmake.configureSettings": {
|
||||||
|
"CMAKE_EXPORT_COMPILE_COMMANDS": "YES",
|
||||||
|
"SIMDJSON_DEVELOPER_MODE": "ON",
|
||||||
|
"SIMDJSON_SINGLEHEADER": "OFF"
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
+17
-7
@@ -3,7 +3,7 @@ cmake_minimum_required(VERSION 3.14)
|
|||||||
project(
|
project(
|
||||||
simdjson
|
simdjson
|
||||||
# The version number is modified by tools/release.py
|
# The version number is modified by tools/release.py
|
||||||
VERSION 3.9.1
|
VERSION 3.10.1
|
||||||
DESCRIPTION "Parsing gigabytes of JSON per second"
|
DESCRIPTION "Parsing gigabytes of JSON per second"
|
||||||
HOMEPAGE_URL "https://simdjson.org/"
|
HOMEPAGE_URL "https://simdjson.org/"
|
||||||
LANGUAGES CXX C
|
LANGUAGES CXX C
|
||||||
@@ -20,10 +20,14 @@ string(
|
|||||||
# ---- Options, variables ----
|
# ---- Options, variables ----
|
||||||
|
|
||||||
# These version numbers are modified by tools/release.py
|
# These version numbers are modified by tools/release.py
|
||||||
set(SIMDJSON_LIB_VERSION "22.0.0" CACHE STRING "simdjson library version")
|
set(SIMDJSON_LIB_VERSION "23.0.0" CACHE STRING "simdjson library version")
|
||||||
set(SIMDJSON_LIB_SOVERSION "22" CACHE STRING "simdjson library soversion")
|
set(SIMDJSON_LIB_SOVERSION "23" CACHE STRING "simdjson library soversion")
|
||||||
|
|
||||||
option(SIMDJSON_BUILD_STATIC_LIB "Build simdjson_static library along with simdjson" OFF)
|
option(SIMDJSON_BUILD_STATIC_LIB "Build simdjson_static library along with simdjson (only makes sense if BUILD_SHARED_LIBS=ON)" OFF)
|
||||||
|
if(SIMDJSON_BUILD_STATIC_LIB AND NOT BUILD_SHARED_LIBS)
|
||||||
|
message(WARNING "SIMDJSON_BUILD_STATIC_LIB only makes sense if BUILD_SHARED_LIBS is set to ON")
|
||||||
|
message(WARNING "You might be building and installing a two identical static libraries.")
|
||||||
|
endif()
|
||||||
|
|
||||||
option(SIMDJSON_ENABLE_THREADS "Link with thread support" ON)
|
option(SIMDJSON_ENABLE_THREADS "Link with thread support" ON)
|
||||||
|
|
||||||
@@ -51,6 +55,7 @@ endif()
|
|||||||
if(is_top_project)
|
if(is_top_project)
|
||||||
option(SIMDJSON_DEVELOPER_MODE "Enable targets for developing simdjson" OFF)
|
option(SIMDJSON_DEVELOPER_MODE "Enable targets for developing simdjson" OFF)
|
||||||
option(BUILD_SHARED_LIBS "Build simdjson as a shared library" OFF)
|
option(BUILD_SHARED_LIBS "Build simdjson as a shared library" OFF)
|
||||||
|
option(SIMDJSON_SINGLEHEADER "Disable singleheader generation" ON)
|
||||||
endif()
|
endif()
|
||||||
|
|
||||||
include(cmake/handle-deprecations.cmake)
|
include(cmake/handle-deprecations.cmake)
|
||||||
@@ -155,11 +160,13 @@ endif()
|
|||||||
include(CMakePackageConfigHelpers)
|
include(CMakePackageConfigHelpers)
|
||||||
include(GNUInstallDirs)
|
include(GNUInstallDirs)
|
||||||
|
|
||||||
install(
|
if(SIMDJSON_SINGLEHEADER)
|
||||||
|
install(
|
||||||
FILES singleheader/simdjson.h
|
FILES singleheader/simdjson.h
|
||||||
DESTINATION "${CMAKE_INSTALL_INCLUDEDIR}"
|
DESTINATION "${CMAKE_INSTALL_INCLUDEDIR}"
|
||||||
COMPONENT simdjson_Development
|
COMPONENT simdjson_Development
|
||||||
)
|
)
|
||||||
|
endif()
|
||||||
|
|
||||||
install(
|
install(
|
||||||
TARGETS simdjson
|
TARGETS simdjson
|
||||||
@@ -203,6 +210,7 @@ if(SIMDJSON_BUILD_STATIC_LIB)
|
|||||||
TARGETS simdjson_static
|
TARGETS simdjson_static
|
||||||
EXPORT simdjson_staticTargets
|
EXPORT simdjson_staticTargets
|
||||||
ARCHIVE COMPONENT simdjson_Development
|
ARCHIVE COMPONENT simdjson_Development
|
||||||
|
INCLUDES DESTINATION "${CMAKE_INSTALL_INCLUDEDIR}"
|
||||||
)
|
)
|
||||||
install(
|
install(
|
||||||
EXPORT simdjson_staticTargets
|
EXPORT simdjson_staticTargets
|
||||||
@@ -279,6 +287,7 @@ enable_testing()
|
|||||||
add_custom_target(all_tests)
|
add_custom_target(all_tests)
|
||||||
|
|
||||||
add_subdirectory(windows)
|
add_subdirectory(windows)
|
||||||
|
include(cmake/CPM.cmake)
|
||||||
add_subdirectory(dependencies) ## This needs to be before tools because of cxxopts
|
add_subdirectory(dependencies) ## This needs to be before tools because of cxxopts
|
||||||
add_subdirectory(tools) ## This needs to be before tests because of cxxopts
|
add_subdirectory(tools) ## This needs to be before tests because of cxxopts
|
||||||
|
|
||||||
@@ -286,8 +295,9 @@ add_subdirectory(tools) ## This needs to be before tests because of cxxopts
|
|||||||
# most of the data has been moved to https://github.com/simdjson/simdjson-data
|
# most of the data has been moved to https://github.com/simdjson/simdjson-data
|
||||||
add_subdirectory(jsonexamples)
|
add_subdirectory(jsonexamples)
|
||||||
|
|
||||||
|
if(SIMDJSON_SINGLEHEADER)
|
||||||
add_subdirectory(singleheader)
|
add_subdirectory(singleheader)
|
||||||
|
endif()
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -38,7 +38,7 @@ PROJECT_NAME = simdjson
|
|||||||
# could be handy for archiving the generated documentation or if some version
|
# could be handy for archiving the generated documentation or if some version
|
||||||
# control system is used.
|
# control system is used.
|
||||||
|
|
||||||
PROJECT_NUMBER = "3.9.1"
|
PROJECT_NUMBER = "3.10.1"
|
||||||
|
|
||||||
# Using the PROJECT_BRIEF tag one can provide an optional one line description
|
# Using the PROJECT_BRIEF tag one can provide an optional one line description
|
||||||
# for a project that appears at the top of each page and should give viewer a
|
# for a project that appears at the top of each page and should give viewer a
|
||||||
|
|||||||
+3
-3
@@ -88,8 +88,8 @@ simdjson's source structure, from the top level, looks like this:
|
|||||||
* simdjson/ondemand.h: the `simdjson::ondemand` namespace. Includes all public ondemand classes.
|
* simdjson/ondemand.h: the `simdjson::ondemand` namespace. Includes all public ondemand classes.
|
||||||
* simdjson/builtin.h: the `simdjson::builtin` namespace. Aliased to the most universal implementation available.
|
* simdjson/builtin.h: the `simdjson::builtin` namespace. Aliased to the most universal implementation available.
|
||||||
* simdjson/builtin/ondemand.h: the `simdjson::builtin::ondemand` namespace.
|
* simdjson/builtin/ondemand.h: the `simdjson::builtin::ondemand` namespace.
|
||||||
* simdjson/arm64|fallback|haswell|icelake|ppc64|westmere/ondemand.h: the `simdjson::<implementation>::ondemand` namespace. on demand compiled for the specific implementation.
|
* simdjson/arm64|fallback|haswell|icelake|ppc64|westmere/ondemand.h: the `simdjson::<implementation>::ondemand` namespace. On-Demand compiled for the specific implementation.
|
||||||
* simdjson/generic/ondemand/*.h: individual on demand classes, generically written.
|
* simdjson/generic/ondemand/*.h: individual On-Demand classes, generically written.
|
||||||
* simdjson/generic/ondemand/dependencies.h: dependencies on common, non-implementation-specific simdjson classes. This will be included before including amalgamated.h.
|
* simdjson/generic/ondemand/dependencies.h: dependencies on common, non-implementation-specific simdjson classes. This will be included before including amalgamated.h.
|
||||||
* simdjson/generic/ondemand/amalgamated.h: all generic ondemand classes for an implementation.
|
* simdjson/generic/ondemand/amalgamated.h: all generic ondemand classes for an implementation.
|
||||||
* **src:** The source files for non-inlined functionality (e.g. the architecture-specific parser
|
* **src:** The source files for non-inlined functionality (e.g. the architecture-specific parser
|
||||||
@@ -99,7 +99,7 @@ simdjson's source structure, from the top level, looks like this:
|
|||||||
* *.cpp: other misc. implementations, such as `simdjson::implementation` and the minifier.
|
* *.cpp: other misc. implementations, such as `simdjson::implementation` and the minifier.
|
||||||
* arm64|fallback|haswell|icelake|ppc64|westmere.cpp: Architecture-specific parser implementations.
|
* arm64|fallback|haswell|icelake|ppc64|westmere.cpp: Architecture-specific parser implementations.
|
||||||
* generic/*.h: `simdjson::<implementation>` namespace. Generic implementation of the parser, particularly the `dom_parser_implementation`.
|
* generic/*.h: `simdjson::<implementation>` namespace. Generic implementation of the parser, particularly the `dom_parser_implementation`.
|
||||||
* generic/stage1/*.h: `simdjson::<implementation>::stage1` namespace. Generic implementation of the simd-heavy tokenizer/indexer pass of the simdjson parser. Used for the On Demand interface
|
* generic/stage1/*.h: `simdjson::<implementation>::stage1` namespace. Generic implementation of the simd-heavy tokenizer/indexer pass of the simdjson parser. Used for the On-Demand interface
|
||||||
* generic/stage2/*.h: `simdjson::<implementation>::stage2` namespace. Generic implementation of the tape creator, which consumes the index from stage 1 and actually parses numbers and string and such. Used for the DOM interface.
|
* generic/stage2/*.h: `simdjson::<implementation>::stage2` namespace. Generic implementation of the tape creator, which consumes the index from stage 1 and actually parses numbers and string and such. Used for the DOM interface.
|
||||||
|
|
||||||
Other important files and directories:
|
Other important files and directories:
|
||||||
|
|||||||
@@ -46,6 +46,7 @@ Real-world usage
|
|||||||
- [Meta Velox](https://velox-lib.io)
|
- [Meta Velox](https://velox-lib.io)
|
||||||
- [Google Pax](https://github.com/google/paxml)
|
- [Google Pax](https://github.com/google/paxml)
|
||||||
- [milvus](https://github.com/milvus-io/milvus)
|
- [milvus](https://github.com/milvus-io/milvus)
|
||||||
|
- [QuestDB](https://questdb.io/blog/questdb-release-8-0-3/)
|
||||||
- [Clang Build Analyzer](https://github.com/aras-p/ClangBuildAnalyzer)
|
- [Clang Build Analyzer](https://github.com/aras-p/ClangBuildAnalyzer)
|
||||||
- [Shopify HeapProfiler](https://github.com/Shopify/heap-profiler)
|
- [Shopify HeapProfiler](https://github.com/Shopify/heap-profiler)
|
||||||
- [StarRocks](https://github.com/StarRocks/starrocks)
|
- [StarRocks](https://github.com/StarRocks/starrocks)
|
||||||
@@ -59,6 +60,7 @@ Real-world usage
|
|||||||
- [vast](https://github.com/tenzir/vast)
|
- [vast](https://github.com/tenzir/vast)
|
||||||
- [ada-url](https://github.com/ada-url/ada)
|
- [ada-url](https://github.com/ada-url/ada)
|
||||||
- [fastgron](https://github.com/adamritter/fastgron)
|
- [fastgron](https://github.com/adamritter/fastgron)
|
||||||
|
- [WasmEdge](https://wasmedge.org)
|
||||||
|
|
||||||
If you are planning to use simdjson in a product, please work from one of our releases.
|
If you are planning to use simdjson in a product, please work from one of our releases.
|
||||||
|
|
||||||
@@ -178,9 +180,9 @@ The simdjson library takes advantage of modern microarchitectures, parallelizing
|
|||||||
instructions, reducing branch misprediction, and reducing data dependency to take advantage of each
|
instructions, reducing branch misprediction, and reducing data dependency to take advantage of each
|
||||||
CPU's multiple execution cores.
|
CPU's multiple execution cores.
|
||||||
|
|
||||||
Our default front-end is called On Demand, and we wrote a paper about it:
|
Our default front-end is called On-Demand, and we wrote a paper about it:
|
||||||
|
|
||||||
- John Keiser, Daniel Lemire, [On-Demand JSON: A Better Way to Parse Documents?](http://arxiv.org/abs/2312.17149), Software: Practice and Experience (to appear)
|
- John Keiser, Daniel Lemire, [On-Demand JSON: A Better Way to Parse Documents?](http://arxiv.org/abs/2312.17149), Software: Practice and Experience 54 (6), 2024.
|
||||||
|
|
||||||
Some people [enjoy reading the first (2019) simdjson paper](https://arxiv.org/abs/1902.08318): A description of the design
|
Some people [enjoy reading the first (2019) simdjson paper](https://arxiv.org/abs/1902.08318): A description of the design
|
||||||
and implementation of simdjson is in our research article:
|
and implementation of simdjson is in our research article:
|
||||||
@@ -199,8 +201,8 @@ For the video inclined, <br />
|
|||||||
Funding
|
Funding
|
||||||
-------
|
-------
|
||||||
|
|
||||||
The work is supported by the Natural Sciences and Engineering Research Council of Canada under grant
|
The work is supported by the Natural Sciences and Engineering Research Council of Canada under grants
|
||||||
number RGPIN-2017-03910.
|
RGPIN-2017-03910 and RGPIN-2024-03787.
|
||||||
|
|
||||||
[license]: LICENSE
|
[license]: LICENSE
|
||||||
[license img]: https://img.shields.io/badge/License-Apache%202-blue.svg
|
[license img]: https://img.shields.io/badge/License-Apache%202-blue.svg
|
||||||
|
|||||||
@@ -50,7 +50,7 @@ private:
|
|||||||
uint8_t *p) noexcept;
|
uint8_t *p) noexcept;
|
||||||
simdjson_inline void
|
simdjson_inline void
|
||||||
write_raw_string(simdjson::ondemand::raw_json_string rjs);
|
write_raw_string(simdjson::ondemand::raw_json_string rjs);
|
||||||
inline void recursive_processor(simdjson::ondemand::value element);
|
inline void recursive_processor(simdjson::ondemand::value&& element);
|
||||||
inline void recursive_processor_ref(simdjson::ondemand::value& element);
|
inline void recursive_processor_ref(simdjson::ondemand::value& element);
|
||||||
|
|
||||||
simdjson::ondemand::parser parser;
|
simdjson::ondemand::parser parser;
|
||||||
@@ -90,12 +90,13 @@ simdjson2msgpack::to_msgpack(const simdjson::padded_string &json,
|
|||||||
} else {
|
} else {
|
||||||
simdjson::ondemand::value val = doc;
|
simdjson::ondemand::value val = doc;
|
||||||
#define SIMDJSON_GCC_COMPILER ((__GNUC__) && !(__clang__) && !(__INTEL_COMPILER))
|
#define SIMDJSON_GCC_COMPILER ((__GNUC__) && !(__clang__) && !(__INTEL_COMPILER))
|
||||||
#if SIMDJSON_GCC_COMPILER
|
#if 1
|
||||||
|
//SIMDJSON_GCC_COMPILER
|
||||||
// the GCC compiler does well with by-value passing.
|
// the GCC compiler does well with by-value passing.
|
||||||
// GCC has superior recursive inlining:
|
// GCC has superior recursive inlining:
|
||||||
// https://stackoverflow.com/questions/29186186/why-does-gcc-generate-a-faster-program-than-clang-in-this-recursive-fibonacci-co
|
// https://stackoverflow.com/questions/29186186/why-does-gcc-generate-a-faster-program-than-clang-in-this-recursive-fibonacci-co
|
||||||
// https://godbolt.org/z/TeK4doE51
|
// https://godbolt.org/z/TeK4doE51
|
||||||
recursive_processor(val);
|
recursive_processor(std::move(val));
|
||||||
#else
|
#else
|
||||||
recursive_processor_ref(val);
|
recursive_processor_ref(val);
|
||||||
#endif
|
#endif
|
||||||
@@ -140,7 +141,7 @@ void simdjson2msgpack::write_raw_string(
|
|||||||
write_uint32_at(uint32_t(v.size()), location);
|
write_uint32_at(uint32_t(v.size()), location);
|
||||||
}
|
}
|
||||||
|
|
||||||
void simdjson2msgpack::recursive_processor(simdjson::ondemand::value element) {
|
void simdjson2msgpack::recursive_processor(simdjson::ondemand::value&& element) {
|
||||||
switch (element.type()) {
|
switch (element.type()) {
|
||||||
case simdjson::ondemand::json_type::array: {
|
case simdjson::ondemand::json_type::array: {
|
||||||
uint32_t counter = 0;
|
uint32_t counter = 0;
|
||||||
@@ -148,7 +149,7 @@ void simdjson2msgpack::recursive_processor(simdjson::ondemand::value element) {
|
|||||||
uint8_t *location = skip_uint32();
|
uint8_t *location = skip_uint32();
|
||||||
for (auto child : element.get_array()) {
|
for (auto child : element.get_array()) {
|
||||||
counter++;
|
counter++;
|
||||||
recursive_processor(child.value());
|
recursive_processor(std::move(child.value()));
|
||||||
}
|
}
|
||||||
write_uint32_at(counter, location);
|
write_uint32_at(counter, location);
|
||||||
} break;
|
} break;
|
||||||
|
|||||||
@@ -13,7 +13,7 @@ class OnDemand {
|
|||||||
public:
|
public:
|
||||||
OnDemand() {
|
OnDemand() {
|
||||||
if(!displayed_implementation) {
|
if(!displayed_implementation) {
|
||||||
std::cout << "On Demand implementation: " << builtin_implementation()->name() << std::endl;
|
std::cout << "On-Demand implementation: " << builtin_implementation()->name() << std::endl;
|
||||||
displayed_implementation = true;
|
displayed_implementation = true;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,24 @@
|
|||||||
|
# SPDX-License-Identifier: MIT
|
||||||
|
#
|
||||||
|
# SPDX-FileCopyrightText: Copyright (c) 2019-2023 Lars Melchior and contributors
|
||||||
|
|
||||||
|
set(CPM_DOWNLOAD_VERSION 0.40.2)
|
||||||
|
set(CPM_HASH_SUM "c8cdc32c03816538ce22781ed72964dc864b2a34a310d3b7104812a5ca2d835d")
|
||||||
|
|
||||||
|
if(CPM_SOURCE_CACHE)
|
||||||
|
set(CPM_DOWNLOAD_LOCATION "${CPM_SOURCE_CACHE}/cpm/CPM_${CPM_DOWNLOAD_VERSION}.cmake")
|
||||||
|
elseif(DEFINED ENV{CPM_SOURCE_CACHE})
|
||||||
|
set(CPM_DOWNLOAD_LOCATION "$ENV{CPM_SOURCE_CACHE}/cpm/CPM_${CPM_DOWNLOAD_VERSION}.cmake")
|
||||||
|
else()
|
||||||
|
set(CPM_DOWNLOAD_LOCATION "${CMAKE_BINARY_DIR}/cmake/CPM_${CPM_DOWNLOAD_VERSION}.cmake")
|
||||||
|
endif()
|
||||||
|
|
||||||
|
# Expand relative path. This is important if the provided path contains a tilde (~)
|
||||||
|
get_filename_component(CPM_DOWNLOAD_LOCATION ${CPM_DOWNLOAD_LOCATION} ABSOLUTE)
|
||||||
|
|
||||||
|
file(DOWNLOAD
|
||||||
|
https://github.com/cpm-cmake/CPM.cmake/releases/download/v${CPM_DOWNLOAD_VERSION}/CPM.cmake
|
||||||
|
${CPM_DOWNLOAD_LOCATION} EXPECTED_HASH SHA256=${CPM_HASH_SUM}
|
||||||
|
)
|
||||||
|
|
||||||
|
include(${CPM_DOWNLOAD_LOCATION})
|
||||||
Vendored
+76
-37
@@ -1,5 +1,4 @@
|
|||||||
include(CMakeDependentOption)
|
include(CMakeDependentOption)
|
||||||
include(import.cmake)
|
|
||||||
|
|
||||||
option(SIMDJSON_ALLOW_DOWNLOADS
|
option(SIMDJSON_ALLOW_DOWNLOADS
|
||||||
"Allow dependencies to be downloaded during configure time"
|
"Allow dependencies to be downloaded during configure time"
|
||||||
@@ -11,17 +10,21 @@ cmake_dependent_option(SIMDJSON_GOOGLE_BENCHMARKS "compile the Google Benchmark
|
|||||||
SIMDJSON_ALLOW_DOWNLOADS OFF)
|
SIMDJSON_ALLOW_DOWNLOADS OFF)
|
||||||
|
|
||||||
if(SIMDJSON_GOOGLE_BENCHMARKS)
|
if(SIMDJSON_GOOGLE_BENCHMARKS)
|
||||||
set_off(BENCHMARK_ENABLE_TESTING)
|
CPMAddPackage(
|
||||||
set_off(BENCHMARK_ENABLE_INSTALL)
|
NAME google_benchmarks
|
||||||
set_off(BENCHMARK_ENABLE_WERROR)
|
URL https://github.com/google/benchmark/archive/refs/tags/v1.7.1.zip
|
||||||
|
OPTIONS
|
||||||
import_dependency(google_benchmarks google/benchmark v1.7.1)
|
"BENCHMARK_ENABLE_TESTING OFF"
|
||||||
add_dependency(google_benchmarks)
|
"BENCHMARK_ENABLE_INSTALL OFF"
|
||||||
|
"BENCHMARK_ENABLE_WERROR OFF"
|
||||||
|
)
|
||||||
endif()
|
endif()
|
||||||
|
|
||||||
# The bulk of our benchmarking and testing data has been moved simdjson/simdjson-data
|
CPMAddPackage(
|
||||||
import_dependency(simdjson-data simdjson/simdjson-data a5b13babe65c1bba7186b41b43d4cbdc20a5c470)
|
NAME simdjson-data
|
||||||
add_dependency(simdjson-data)
|
URL https://github.com/simdjson/simdjson-data/archive/a5b13babe65c1bba7186b41b43d4cbdc20a5c470.zip
|
||||||
|
)
|
||||||
|
|
||||||
option(SIMDJSON_USE_BOOST_JSON "Try to include BOOST_JSON, this may break your binaries under some systems." OFF)
|
option(SIMDJSON_USE_BOOST_JSON "Try to include BOOST_JSON, this may break your binaries under some systems." OFF)
|
||||||
# This prevents variables declared with set() from unnecessarily escaping and
|
# This prevents variables declared with set() from unnecessarily escaping and
|
||||||
# should not be called more than once
|
# should not be called more than once
|
||||||
@@ -38,20 +41,30 @@ function(competition_scope_)
|
|||||||
int main() {}
|
int main() {}
|
||||||
]] SIMDJSON_FOUND_STRING_VIEW)
|
]] SIMDJSON_FOUND_STRING_VIEW)
|
||||||
if(SIMDJSON_FOUND_STRING_VIEW AND SIMDJSON_USE_BOOST_JSON)
|
if(SIMDJSON_FOUND_STRING_VIEW AND SIMDJSON_USE_BOOST_JSON)
|
||||||
import_dependency(boostjson boostorg/json ee8d72d)
|
CPMAddPackage(
|
||||||
|
NAME boostjson
|
||||||
|
URL https://github.com/boostorg/json/archive/ee8d72d8502b409b5561200299cad30ccdb91415.zip
|
||||||
|
)
|
||||||
add_library(boostjson STATIC "${boostjson_SOURCE_DIR}/src/src.cpp")
|
add_library(boostjson STATIC "${boostjson_SOURCE_DIR}/src/src.cpp")
|
||||||
target_compile_definitions(boostjson PUBLIC BOOST_JSON_STANDALONE)
|
target_compile_definitions(boostjson PUBLIC BOOST_JSON_STANDALONE)
|
||||||
target_include_directories(boostjson SYSTEM PUBLIC
|
target_include_directories(boostjson SYSTEM PUBLIC
|
||||||
"${boostjson_SOURCE_DIR}/include")
|
"${boostjson_SOURCE_DIR}/include")
|
||||||
target_compile_definitions(boostjson INTERFACE SIMDJSON_COMPETITION_BOOSTJSON)
|
target_compile_definitions(boostjson INTERFACE SIMDJSON_COMPETITION_BOOSTJSON)
|
||||||
endif()
|
endif()
|
||||||
|
CPMAddPackage(
|
||||||
import_dependency(cjson DaveGamble/cJSON c69134d)
|
NAME cjson
|
||||||
|
URL https://github.com/DaveGamble/cJSON/archive/c69134d01746dcf551dd7724b4edb12f922eb0d1.zip
|
||||||
|
DOWNLOAD_ONLY YES
|
||||||
|
)
|
||||||
add_library(cjson STATIC "${cjson_SOURCE_DIR}/cJSON.c")
|
add_library(cjson STATIC "${cjson_SOURCE_DIR}/cJSON.c")
|
||||||
target_include_directories(cjson SYSTEM PUBLIC "${cjson_SOURCE_DIR}")
|
target_include_directories(cjson SYSTEM PUBLIC "${cjson_SOURCE_DIR}")
|
||||||
target_compile_definitions(cjson INTERFACE SIMDJSON_COMPETITION_CJSON)
|
target_compile_definitions(cjson INTERFACE SIMDJSON_COMPETITION_CJSON)
|
||||||
|
|
||||||
import_dependency(fastjson mikeando/fastjson 485f994)
|
CPMAddPackage(
|
||||||
|
NAME fastjson
|
||||||
|
URL https://github.com/mikeando/fastjson/archive/485f994a61a64ac73fa6a40d4d639b99b463563b.zip
|
||||||
|
DOWNLOAD_ONLY YES
|
||||||
|
)
|
||||||
add_library(fastjson STATIC
|
add_library(fastjson STATIC
|
||||||
"${fastjson_SOURCE_DIR}/src/fastjson.cpp"
|
"${fastjson_SOURCE_DIR}/src/fastjson.cpp"
|
||||||
"${fastjson_SOURCE_DIR}/src/fastjson2.cpp"
|
"${fastjson_SOURCE_DIR}/src/fastjson2.cpp"
|
||||||
@@ -60,28 +73,36 @@ int main() {}
|
|||||||
"${fastjson_SOURCE_DIR}/include")
|
"${fastjson_SOURCE_DIR}/include")
|
||||||
target_compile_definitions(fastjson INTERFACE SIMDJSON_COMPETITION_FASTJSON)
|
target_compile_definitions(fastjson INTERFACE SIMDJSON_COMPETITION_FASTJSON)
|
||||||
|
|
||||||
import_dependency(gason vivkin/gason 7aee524)
|
CPMAddPackage(
|
||||||
|
NAME gason
|
||||||
|
URL https://github.com/vivkin/gason/archive/7aee524189da1c1ecd19f67981e3d903dae25470.zip
|
||||||
|
DOWNLOAD_ONLY YES
|
||||||
|
)
|
||||||
add_library(gason STATIC "${gason_SOURCE_DIR}/src/gason.cpp")
|
add_library(gason STATIC "${gason_SOURCE_DIR}/src/gason.cpp")
|
||||||
target_include_directories(gason SYSTEM PUBLIC "${gason_SOURCE_DIR}/src")
|
target_include_directories(gason SYSTEM PUBLIC "${gason_SOURCE_DIR}/src")
|
||||||
target_compile_definitions(gason INTERFACE SIMDJSON_COMPETITION_GASON)
|
target_compile_definitions(gason INTERFACE SIMDJSON_COMPETITION_GASON)
|
||||||
|
|
||||||
import_dependency(jsmn zserge/jsmn 18e9fe4)
|
CPMAddPackage(
|
||||||
|
NAME jsmn
|
||||||
|
URL https://github.com/zserge/jsmn/archive/18e9fe42cbfe21d65076f5c77ae2be379ad1270f.zip
|
||||||
|
DOWNLOAD_ONLY YES
|
||||||
|
)
|
||||||
add_library(jsmn STATIC "${jsmn_SOURCE_DIR}/jsmn.c")
|
add_library(jsmn STATIC "${jsmn_SOURCE_DIR}/jsmn.c")
|
||||||
target_include_directories(jsmn SYSTEM PUBLIC "${jsmn_SOURCE_DIR}")
|
target_include_directories(jsmn SYSTEM PUBLIC "${jsmn_SOURCE_DIR}")
|
||||||
target_compile_definitions(jsmn INTERFACE SIMDJSON_COMPETITION_JSMN)
|
target_compile_definitions(jsmn INTERFACE SIMDJSON_COMPETITION_JSMN)
|
||||||
|
|
||||||
message(STATUS "Importing json (nlohmann/json@v3.10.5)")
|
CPMAddPackage(
|
||||||
set(nlohmann_json_SOURCE_DIR "${dep_root}/json")
|
NAME nlohmann_json
|
||||||
if(NOT EXISTS "${nlohmann_json_SOURCE_DIR}")
|
URL https://github.com/nlohmann/json/archive/refs/tags/v3.10.5.zip
|
||||||
file(DOWNLOAD
|
)
|
||||||
"https://github.com/nlohmann/json/releases/download/v3.10.5/json.hpp"
|
|
||||||
"${nlohmann_json_SOURCE_DIR}/nlohmann/json.hpp")
|
|
||||||
endif()
|
|
||||||
add_library(nlohmann_json INTERFACE)
|
|
||||||
target_include_directories(nlohmann_json SYSTEM INTERFACE "${nlohmann_json_SOURCE_DIR}")
|
|
||||||
target_compile_definitions(nlohmann_json INTERFACE SIMDJSON_COMPETITION_NLOHMANN_JSON)
|
|
||||||
|
|
||||||
import_dependency(json11 dropbox/json11 ec4e452)
|
set_property(TARGET nlohmann_json APPEND PROPERTY INTERFACE_COMPILE_DEFINITIONS SIMDJSON_COMPETITION_NLOHMANN_JSON)
|
||||||
|
|
||||||
|
CPMAddPackage(
|
||||||
|
NAME json11
|
||||||
|
URL https://github.com/dropbox/json11/archive/ec4e45219af1d7cde3d58b49ed762376fccf1ace.zip
|
||||||
|
DOWNLOAD_ONLY YES
|
||||||
|
)
|
||||||
add_library(json11 STATIC "${json11_SOURCE_DIR}/json11.cpp")
|
add_library(json11 STATIC "${json11_SOURCE_DIR}/json11.cpp")
|
||||||
target_include_directories(json11 SYSTEM PUBLIC "${json11_SOURCE_DIR}")
|
target_include_directories(json11 SYSTEM PUBLIC "${json11_SOURCE_DIR}")
|
||||||
target_compile_definitions(json11 INTERFACE SIMDJSON_COMPETITION_JSON11)
|
target_compile_definitions(json11 INTERFACE SIMDJSON_COMPETITION_JSON11)
|
||||||
@@ -91,7 +112,11 @@ int main() {}
|
|||||||
target_include_directories(jsoncpp SYSTEM PUBLIC "${jsoncpp_SOURCE_DIR}")
|
target_include_directories(jsoncpp SYSTEM PUBLIC "${jsoncpp_SOURCE_DIR}")
|
||||||
target_compile_definitions(jsoncpp INTERFACE SIMDJSON_COMPETITION_JSONCPP)
|
target_compile_definitions(jsoncpp INTERFACE SIMDJSON_COMPETITION_JSONCPP)
|
||||||
|
|
||||||
import_dependency(rapidjson Tencent/rapidjson f54b0e4)
|
CPMAddPackage(
|
||||||
|
NAME rapidjson
|
||||||
|
URL https://github.com/Tencent/rapidjson/archive/f54b0e47a08782a6131cc3d60f94d038fa6e0a51.zip
|
||||||
|
DOWNLOAD_ONLY YES
|
||||||
|
)
|
||||||
add_library(rapidjson INTERFACE)
|
add_library(rapidjson INTERFACE)
|
||||||
target_compile_definitions(rapidjson INTERFACE RAPIDJSON_HAS_STDSTRING)
|
target_compile_definitions(rapidjson INTERFACE RAPIDJSON_HAS_STDSTRING)
|
||||||
include (TestBigEndian)
|
include (TestBigEndian)
|
||||||
@@ -110,14 +135,22 @@ int main() {}
|
|||||||
target_compile_definitions(rapidjson INTERFACE SIMDJSON_COMPETITION_RAPIDJSON)
|
target_compile_definitions(rapidjson INTERFACE SIMDJSON_COMPETITION_RAPIDJSON)
|
||||||
endif()
|
endif()
|
||||||
|
|
||||||
import_dependency(sajson chadaustin/sajson 2dcfd35)
|
CPMAddPackage(
|
||||||
|
NAME sajson
|
||||||
|
URL https://github.com/chadaustin/sajson/archive/2dcfd350586375f9910f74821d4f07d67ae455ba.zip
|
||||||
|
DOWNLOAD_ONLY YES
|
||||||
|
)
|
||||||
add_library(sajson INTERFACE)
|
add_library(sajson INTERFACE)
|
||||||
target_compile_definitions(sajson INTERFACE SAJSON_UNSORTED_OBJECT_KEYS)
|
target_compile_definitions(sajson INTERFACE SAJSON_UNSORTED_OBJECT_KEYS)
|
||||||
target_include_directories(sajson SYSTEM INTERFACE
|
target_include_directories(sajson SYSTEM INTERFACE
|
||||||
"${sajson_SOURCE_DIR}/include")
|
"${sajson_SOURCE_DIR}/include")
|
||||||
target_compile_definitions(sajson INTERFACE SIMDJSON_COMPETITION_SAJSON)
|
target_compile_definitions(sajson INTERFACE SIMDJSON_COMPETITION_SAJSON)
|
||||||
|
|
||||||
import_dependency(ujson4c esnme/ujson4c e14f3fd)
|
CPMAddPackage(
|
||||||
|
NAME ujson4c
|
||||||
|
URL https://github.com/esnme/ujson4c/archive/e14f3fd5207fe30d1bdea723f260609e69d1abfa.zip
|
||||||
|
DOWNLOAD_ONLY YES
|
||||||
|
)
|
||||||
add_library(ujson4c STATIC
|
add_library(ujson4c STATIC
|
||||||
"${ujson4c_SOURCE_DIR}/src/ujdecode.c"
|
"${ujson4c_SOURCE_DIR}/src/ujdecode.c"
|
||||||
"${ujson4c_SOURCE_DIR}/3rdparty/ultrajsondec.c")
|
"${ujson4c_SOURCE_DIR}/3rdparty/ultrajsondec.c")
|
||||||
@@ -126,7 +159,11 @@ int main() {}
|
|||||||
"${ujson4c_SOURCE_DIR}/3rdparty")
|
"${ujson4c_SOURCE_DIR}/3rdparty")
|
||||||
target_compile_definitions(ujson4c INTERFACE SIMDJSON_COMPETITION_UJSON4C)
|
target_compile_definitions(ujson4c INTERFACE SIMDJSON_COMPETITION_UJSON4C)
|
||||||
|
|
||||||
import_dependency(yyjson ibireme/yyjson c385651)
|
CPMAddPackage(
|
||||||
|
NAME yyjson
|
||||||
|
URL https://github.com/ibireme/yyjson/archive/c3856514de0a67d7b66939bf3ed491a2d6e61277.zip
|
||||||
|
DOWNLOAD_ONLY YES
|
||||||
|
)
|
||||||
add_library(yyjson STATIC "${yyjson_SOURCE_DIR}/src/yyjson.c")
|
add_library(yyjson STATIC "${yyjson_SOURCE_DIR}/src/yyjson.c")
|
||||||
target_include_directories(yyjson SYSTEM PUBLIC "${yyjson_SOURCE_DIR}/src")
|
target_include_directories(yyjson SYSTEM PUBLIC "${yyjson_SOURCE_DIR}/src")
|
||||||
target_compile_definitions(yyjson INTERFACE SIMDJSON_COMPETITION_YYJSON)
|
target_compile_definitions(yyjson INTERFACE SIMDJSON_COMPETITION_YYJSON)
|
||||||
@@ -151,10 +188,12 @@ cmake_dependent_option(SIMDJSON_CXXOPTS "Download cxxopts (necessary for tools)"
|
|||||||
SIMDJSON_ALLOW_DOWNLOADS OFF)
|
SIMDJSON_ALLOW_DOWNLOADS OFF)
|
||||||
|
|
||||||
if(SIMDJSON_CXXOPTS)
|
if(SIMDJSON_CXXOPTS)
|
||||||
set_off(CXXOPTS_BUILD_EXAMPLES)
|
CPMAddPackage(
|
||||||
set_off(CXXOPTS_BUILD_TESTS)
|
NAME cxxopts
|
||||||
set_off(CXXOPTS_ENABLE_INSTALL)
|
URL https://github.com/jarro2783/cxxopts/archive/59656709c0c58fcd0ed18b38e02938dbe05284c5.zip
|
||||||
|
OPTIONS
|
||||||
import_dependency(cxxopts jarro2783/cxxopts 794c975)
|
"CXXOPTS_BUILD_EXAMPLES OFF"
|
||||||
add_dependency(cxxopts)
|
"CXXOPTS_BUILD_TESTS OFF"
|
||||||
|
"CXXOPTS_ENABLE_INSTALL OFF"
|
||||||
|
)
|
||||||
endif()
|
endif()
|
||||||
|
|||||||
Vendored
-48
@@ -1,48 +0,0 @@
|
|||||||
set(dep_root "${simdjson_SOURCE_DIR}/dependencies/.cache")
|
|
||||||
if(DEFINED ENV{simdjson_DEPENDENCY_CACHE_DIR})
|
|
||||||
set(dep_root "$ENV{simdjson_DEPENDENCY_CACHE_DIR}")
|
|
||||||
endif()
|
|
||||||
|
|
||||||
function(import_dependency NAME GITHUB_REPO COMMIT)
|
|
||||||
message(STATUS "Importing ${NAME} (${GITHUB_REPO}@${COMMIT})")
|
|
||||||
set(target "${dep_root}/${NAME}")
|
|
||||||
|
|
||||||
# If the folder exists in the cache, then we assume that everything is as
|
|
||||||
# should be and do nothing
|
|
||||||
if(EXISTS "${target}")
|
|
||||||
set("${NAME}_SOURCE_DIR" "${target}" PARENT_SCOPE)
|
|
||||||
return()
|
|
||||||
endif()
|
|
||||||
|
|
||||||
set(zip_url "https://github.com/${GITHUB_REPO}/archive/${COMMIT}.zip")
|
|
||||||
set(archive "${dep_root}/archive.zip")
|
|
||||||
set(dest "${dep_root}/_extract")
|
|
||||||
|
|
||||||
file(DOWNLOAD "${zip_url}" "${archive}")
|
|
||||||
file(MAKE_DIRECTORY "${dest}")
|
|
||||||
execute_process(
|
|
||||||
WORKING_DIRECTORY "${dest}"
|
|
||||||
COMMAND "${CMAKE_COMMAND}" -E tar xf "${archive}")
|
|
||||||
file(REMOVE "${archive}")
|
|
||||||
|
|
||||||
# GitHub archives only ever have one folder component at the root, so this
|
|
||||||
# will always match that single folder
|
|
||||||
file(GLOB dir LIST_DIRECTORIES YES "${dest}/*")
|
|
||||||
|
|
||||||
file(RENAME "${dir}" "${target}")
|
|
||||||
|
|
||||||
set("${NAME}_SOURCE_DIR" "${target}" PARENT_SCOPE)
|
|
||||||
endfunction()
|
|
||||||
|
|
||||||
# Delegates to the dependency
|
|
||||||
macro(add_dependency NAME)
|
|
||||||
if(NOT DEFINED "${NAME}_SOURCE_DIR")
|
|
||||||
message(FATAL_ERROR "Missing ${NAME}_SOURCE_DIR variable")
|
|
||||||
endif()
|
|
||||||
|
|
||||||
add_subdirectory("${${NAME}_SOURCE_DIR}" "${PROJECT_BINARY_DIR}/_deps/${NAME}" EXCLUDE_FROM_ALL)
|
|
||||||
endmacro()
|
|
||||||
|
|
||||||
function(set_off NAME)
|
|
||||||
set("${NAME}" OFF CACHE INTERNAL "")
|
|
||||||
endfunction()
|
|
||||||
+184
-84
@@ -9,36 +9,37 @@ An overview of what you need to know to use simdjson, with examples.
|
|||||||
- [Using simdjson with package managers](#using-simdjson-with-package-managers)
|
- [Using simdjson with package managers](#using-simdjson-with-package-managers)
|
||||||
- [Using simdjson as a CMake dependency](#using-simdjson-as-a-cmake-dependency)
|
- [Using simdjson as a CMake dependency](#using-simdjson-as-a-cmake-dependency)
|
||||||
- [Versions](#versions)
|
- [Versions](#versions)
|
||||||
- [The Basics: Loading and Parsing JSON Documents](#the-basics-loading-and-parsing-json-documents)
|
- [The basics: loading and parsing JSON documents](#the-basics-loading-and-parsing-json-documents)
|
||||||
- [Documents are Iterators](#documents-are-iterators)
|
- [Documents are iterators](#documents-are-iterators)
|
||||||
- [Parser, Document and JSON Scope](#parser-document-and-json-scope)
|
- [Parser, document and JSON scope](#parser-document-and-json-scope)
|
||||||
- [string_view](#string_view)
|
- [string_view](#string_view)
|
||||||
- [Using the Parsed JSON](#using-the-parsed-json)
|
- [Avoiding pitfalls: enable development checks](#avoiding-pitfalls-enable-development-checks)
|
||||||
- [Using the Parsed JSON: Additional examples](#using-the-parsed-json-additional-examples)
|
- [Using the parsed JSON](#using-the-parsed-json)
|
||||||
|
- [Using the parsed JSON: additional examples](#using-the-parsed-json-additional-examples)
|
||||||
- [Adding support for custom types](#adding-support-for-custom-types)
|
- [Adding support for custom types](#adding-support-for-custom-types)
|
||||||
- [Minifying JSON strings without parsing](#minifying-json-strings-without-parsing)
|
- [Minifying JSON strings without parsing](#minifying-json-strings-without-parsing)
|
||||||
- [UTF-8 validation (alone)](#utf-8-validation-alone)
|
- [UTF-8 validation (alone)](#utf-8-validation-alone)
|
||||||
- [JSON Pointer](#json-pointer)
|
- [JSON Pointer](#json-pointer)
|
||||||
- [JSONPath](#json-path)
|
- [JSONPath](#jsonpath)
|
||||||
- [Error Handling](#error-handling)
|
- [Error handling](#error-handling)
|
||||||
- [Error Handling Examples without Exceptions](#error-handling-examples-without-exceptions)
|
- [Error handling examples without exceptions](#error-handling-examples-without-exceptions)
|
||||||
- [Disabling Exceptions](#disabling-exceptions)
|
- [Disabling exceptions](#disabling-exceptions)
|
||||||
- [Exceptions](#exceptions)
|
- [Exceptions](#exceptions)
|
||||||
- [Current location in document](#current-location-in-document)
|
- [Current location in document](#current-location-in-document)
|
||||||
- [Checking for trailing content](#checking-for-trailing-content)
|
- [Checking for trailing content](#checking-for-trailing-content)
|
||||||
- [Rewinding](#rewinding)
|
- [Rewinding](#rewinding)
|
||||||
- [Newline-Delimited JSON (ndjson) and JSON lines](#newline-delimited-json-ndjson-and-json-lines)
|
- [Newline-Delimited JSON (ndjson) and JSON lines](#newline-delimited-json-ndjson-and-json-lines)
|
||||||
- [Parsing Numbers Inside Strings](#parsing-numbers-inside-strings)
|
- [Parsing numbers inside strings](#parsing-numbers-inside-strings)
|
||||||
- [Dynamic Number Types](#dynamic-number-types)
|
- [Dynamic Number Types](#dynamic-number-types)
|
||||||
- [Raw Strings From Keys](#raw-strings-from-keys)
|
- [Raw strings from keys](#raw-strings-from-keys)
|
||||||
- [General Direct Access to the Raw JSON String](#general-direct-access-to-the-raw-json-string)
|
- [General direct access to the raw JSON string](#general-direct-access-to-the-raw-json-string)
|
||||||
- [Storing Directly into an Existing String Instance](#storing-directly-into-an-existing-string-instance)
|
- [Storing directly into an existing string instance](#storing-directly-into-an-existing-string-instance)
|
||||||
- [Thread Safety](#thread-safety)
|
- [Thread safety](#thread-safety)
|
||||||
- [Standard Compliance](#standard-compliance)
|
- [Standard compliance](#standard-compliance)
|
||||||
- [Backwards Compatibility](#backwards-compatibility)
|
- [Backwards compatibility](#backwards-compatibility)
|
||||||
- [Examples](#examples)
|
- [Examples](#examples)
|
||||||
- [Performance Tips](#performance-tips)
|
- [Performance tips](#performance-tips)
|
||||||
- [Further Reading](#further-reading)
|
- [Further reading](#further-reading)
|
||||||
|
|
||||||
|
|
||||||
Requirements
|
Requirements
|
||||||
@@ -50,7 +51,7 @@ Requirements
|
|||||||
|
|
||||||
Support for AVX-512 require a processor with AVX512-VBMI2 support (Ice Lake or better, AMD Zen 4 or better) under a 64-bit system and a recent compiler (LLVM clang 6 or better, GCC 8 or better, Visual Studio 2019 or better). You need a correspondingly recent assembler such as gas (2.30+) or nasm (2.14+): recent compilers usually come with recent assemblers. If you mix a recent compiler with an incompatible/old assembler (e.g., when using a recent compiler with an old Linux distribution), you may get errors at build time because the compiler produces instructions that the assembler does not recognize: you should update your assembler to match your compiler (e.g., upgrade binutils to version 2.30 or better under Linux) or use an older compiler matching the capabilities of your assembler.
|
Support for AVX-512 require a processor with AVX512-VBMI2 support (Ice Lake or better, AMD Zen 4 or better) under a 64-bit system and a recent compiler (LLVM clang 6 or better, GCC 8 or better, Visual Studio 2019 or better). You need a correspondingly recent assembler such as gas (2.30+) or nasm (2.14+): recent compilers usually come with recent assemblers. If you mix a recent compiler with an incompatible/old assembler (e.g., when using a recent compiler with an old Linux distribution), you may get errors at build time because the compiler produces instructions that the assembler does not recognize: you should update your assembler to match your compiler (e.g., upgrade binutils to version 2.30 or better under Linux) or use an older compiler matching the capabilities of your assembler.
|
||||||
|
|
||||||
We test the library on a big-endian system (IBM s390x with Linux) .
|
We test the library on a big-endian system (IBM s390x with Linux).
|
||||||
|
|
||||||
Including simdjson
|
Including simdjson
|
||||||
------------------
|
------------------
|
||||||
@@ -63,20 +64,25 @@ into your project. Then include it in your project with:
|
|||||||
using namespace simdjson; // optional
|
using namespace simdjson; // optional
|
||||||
```
|
```
|
||||||
|
|
||||||
You can compile with:
|
Under most systems, you can compile with:
|
||||||
|
|
||||||
```
|
```
|
||||||
c++ myproject.cpp simdjson.cpp
|
c++ myproject.cpp simdjson.cpp
|
||||||
```
|
```
|
||||||
|
|
||||||
Note:
|
Note:
|
||||||
|
- We recommend that you use simdjson by copying the single-header `simdjson.h` file along with the source file `simdjson.cpp` directly in your project, as they are part of [every release](https://github.com/simdjson/simdjson/releases) as assets. In this manner, you only have to compile `simdjson.cpp` as any other source file: it works well in every development environment. However, you may also use simdjson as a git submodule ([example](https://github.com/simdjson/cmakedemo)), using FetchContent ([example](https://github.com/simdjson/cmake_demo_single_file)), with ExternalProject_Add ([example](https://github.com/simdjson/cmakedemo_externalproject)) or with CPM ([example](https://github.com/cpm-cmake/CPM.cmake/tree/master/examples/simdjson)).
|
||||||
- Users on macOS and other platforms where default compilers do not provide C++11 compliant by default should request it with the appropriate flag (e.g., `c++ -std=c++11 myproject.cpp simdjson.cpp`).
|
- Users on macOS and other platforms where default compilers do not provide C++11 compliant by default should request it with the appropriate flag (e.g., `c++ -std=c++11 myproject.cpp simdjson.cpp`).
|
||||||
- The library relies on [runtime CPU detection](implementation-selection.md): avoid specifying an architecture at compile time (e.g., `-march-native`) if you want your binaries to run everywhere.
|
- The library relies on [runtime CPU detection](implementation-selection.md): avoid specifying an architecture at compile time (e.g., `-march-native`) if you want your binaries to run everywhere.
|
||||||
|
|
||||||
Using simdjson with package managers
|
Using simdjson with package managers
|
||||||
------------------
|
------------------
|
||||||
|
|
||||||
You can install the simdjson library on your system or in your project using multiple package managers such as MSYS2, the conan package manager, vcpkg, brew, the apt package manager (debian-based Linux systems), the FreeBSD package manager (FreeBSD), and so on. [Visit our wiki for more details](https://github.com/simdjson/simdjson/wiki/Installing-simdjson-with-a-package-manager).
|
You can install the simdjson library on your system or in your project using multiple package managers such as MSYS2, the conan package manager, vcpkg, brew, the apt package manager (debian-based Linux systems), the FreeBSD package manager (FreeBSD), and so on. E.g., [we provide an complete example with vcpkg](https://github.com/simdjson/simdjson-vcpkg) that works under Windows. [Visit our wiki for more details](https://github.com/simdjson/simdjson/wiki/Installing-simdjson-with-a-package-manager).
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
The following Linux distributions provide simdjson packages: Alpine, RedHat, Rocky Linux, Debian, Fedora, and Ubuntu.
|
||||||
|
|
||||||
Using simdjson as a CMake dependency
|
Using simdjson as a CMake dependency
|
||||||
------------------
|
------------------
|
||||||
@@ -141,7 +147,7 @@ https://github.com/simdjson/simdjson/blob/vx.y.z/doc/basics.md
|
|||||||
where `x.y.z` should correspond to the version number you have
|
where `x.y.z` should correspond to the version number you have
|
||||||
chosen.
|
chosen.
|
||||||
|
|
||||||
The Basics: Loading and Parsing JSON Documents
|
The basics: loading and parsing JSON documents
|
||||||
----------------------------------------------
|
----------------------------------------------
|
||||||
|
|
||||||
The simdjson library allows you to navigate and validate JSON documents ([RFC 8259](https://www.tbray.org/ongoing/When/201x/2017/12/14/rfc8259.html)).
|
The simdjson library allows you to navigate and validate JSON documents ([RFC 8259](https://www.tbray.org/ongoing/When/201x/2017/12/14/rfc8259.html)).
|
||||||
@@ -153,7 +159,8 @@ For efficiency reasons, simdjson requires a string with a few bytes (`simdjson::
|
|||||||
at the end, these bytes may be read but their content does not affect the parsing. In practice,
|
at the end, these bytes may be read but their content does not affect the parsing. In practice,
|
||||||
it means that the JSON inputs should be stored in a memory region with `simdjson::SIMDJSON_PADDING`
|
it means that the JSON inputs should be stored in a memory region with `simdjson::SIMDJSON_PADDING`
|
||||||
extra bytes at the end. You do not have to set these bytes to specific values though you may
|
extra bytes at the end. You do not have to set these bytes to specific values though you may
|
||||||
want to if you want to avoid runtime warnings with some sanitizers.
|
want to if you want to avoid runtime warnings with some sanitizers. Advanced users may want to
|
||||||
|
read the section Free Padding in [our performance notes](performance.md).
|
||||||
|
|
||||||
The simdjson library offers a tree-like [API](https://en.wikipedia.org/wiki/API), which you can
|
The simdjson library offers a tree-like [API](https://en.wikipedia.org/wiki/API), which you can
|
||||||
access by creating a `ondemand::parser` and calling the `iterate()` method. The iterate method
|
access by creating a `ondemand::parser` and calling the `iterate()` method. The iterate method
|
||||||
@@ -204,7 +211,7 @@ simdjson::padded_string my_padded_data(data); // copies to a padded buffer
|
|||||||
We recommend against creating many `std::string` or many `std::padding_string` instances in your application to store your JSON data.
|
We recommend against creating many `std::string` or many `std::padding_string` instances in your application to store your JSON data.
|
||||||
Consider reusing the same buffers and limiting memory allocations.
|
Consider reusing the same buffers and limiting memory allocations.
|
||||||
|
|
||||||
By default, the simdjson library throws exceptions (`simdjson_error`) on errors. We omit `try`-`catch` clauses from our illustrating examples: if you omit `try`-`catch` in your code, an uncaught exception will halt your program. It is also possible to use simdjson without generating exceptions, and you may even build the library without exception support at all. See [Error Handling](#error-handling) for details.
|
By default, the simdjson library throws exceptions (`simdjson_error`) on errors. We omit `try`-`catch` clauses from our illustrating examples: if you omit `try`-`catch` in your code, an uncaught exception will halt your program. It is also possible to use simdjson without generating exceptions, and you may even build the library without exception support at all. See [Error handling](#error-handling) for details.
|
||||||
|
|
||||||
|
|
||||||
Some users may want to browse code along with the compiled assembly. You want to check out the following lists of examples:
|
Some users may want to browse code along with the compiled assembly. You want to check out the following lists of examples:
|
||||||
@@ -219,28 +226,28 @@ Further, they may use the AreFileApisANSI function to determine whether
|
|||||||
the filename is interpreted using the ANSI or the system default OEM
|
the filename is interpreted using the ANSI or the system default OEM
|
||||||
codepage, and they may call SetFileApisToOEM accordingly.
|
codepage, and they may call SetFileApisToOEM accordingly.
|
||||||
|
|
||||||
Documents are Iterators
|
Documents are iterators
|
||||||
-----------------------
|
-----------------------
|
||||||
|
|
||||||
The simdjson library relies on an approach to parsing JSON that we call "On Demand".
|
The simdjson library relies on an approach to parsing JSON that we call "On-Demand".
|
||||||
A `document` is *not* a fully-parsed JSON value; rather, it is an **iterator** over the JSON text.
|
A `document` is *not* a fully-parsed JSON value; rather, it is an **iterator** over the JSON text.
|
||||||
This means that while you iterate an array, or search for a field in an object, it is actually
|
This means that while you iterate an array, or search for a field in an object, it is actually
|
||||||
walking through the original JSON text, merrily reading commas and colons and brackets to make sure
|
walking through the original JSON text, merrily reading commas and colons and brackets to make sure
|
||||||
you get where you are going. This is the key to On Demand's performance: since it's just an iterator,
|
you get where you are going. This is the key to On-Demand's performance: since it's just an iterator,
|
||||||
it lets you parse values as you use them. And particularly, it lets you *skip* values you do not want
|
it lets you parse values as you use them. And particularly, it lets you *skip* values you do not want
|
||||||
to use. On Demand is also ideally suited when you want to capture part of the document without parsing it
|
to use. On-Demand is also ideally suited when you want to capture part of the document without parsing it
|
||||||
immediately (e.g., see [General Direct Access to the Raw JSON String](#general-direct-access-to-the-raw-json-string)).
|
immediately (e.g., see [General direct access to the raw JSON string](#general-direct-access-to-the-raw-json-string)).
|
||||||
|
|
||||||
We refer to "On Demand" as a front-end component since it is an interface between the
|
We refer to "On-Demand" as a front-end component since it is an interface between the
|
||||||
low-level parsing functions and the user. It hides much of the complexity of parsing JSON
|
low-level parsing functions and the user. It hides much of the complexity of parsing JSON
|
||||||
documents.
|
documents.
|
||||||
|
|
||||||
### Parser, Document and JSON Scope
|
### Parser, document and JSON scope
|
||||||
|
|
||||||
For code safety, you should keep (1) the `parser` instance, (2) the input string and (3) the document instance alive throughout your parsing. Additionally, you should follow the following rules:
|
For code safety, you should keep (1) the `parser` instance, (2) the input string and (3) the document instance alive throughout your parsing. Additionally, you should follow the following rules:
|
||||||
|
|
||||||
- A `parser` may have at most one document open at a time, since it holds allocated memory used for the parsing.
|
- A `parser` may have at most one document open at a time, since it holds allocated memory used for the parsing.
|
||||||
- By design, you should only have one `document` instance per JSON document. Thus, if you must pass a document instance to a function, you should avoid passing it by value: choose to pass it by reference instance to avoid the copy. (We also provide a `document_reference` class if you need to pass by value.)
|
- By design, you should only have one `document` instance per JSON document. Thus, if you must pass a document instance to a function, you should avoid passing it by value: choose to pass it by reference instance to avoid the copy. In any case, the `document` class does not have a copy constructor.
|
||||||
|
|
||||||
During the `iterate` call, the original JSON text is never modified--only read. After you are done
|
During the `iterate` call, the original JSON text is never modified--only read. After you are done
|
||||||
with the document, the source (whether file or string) can be safely discarded.
|
with the document, the source (whether file or string) can be safely discarded.
|
||||||
@@ -264,7 +271,7 @@ copy the data into their own favorite class instances (e.g., alternatives to `st
|
|||||||
|
|
||||||
A `std::string_view` instance is effectively just a pointer to a region in memory representing
|
A `std::string_view` instance is effectively just a pointer to a region in memory representing
|
||||||
a string. In simdjson, we return `std::string_view` instances that either point within the
|
a string. In simdjson, we return `std::string_view` instances that either point within the
|
||||||
input string you parsed (see [General Direct Access to the Raw JSON String](#general-direct-access-to-the-raw-json-string)), or to a temporary string buffer inside
|
input string you parsed (see [General direct access to the raw JSON string](#general-direct-access-to-the-raw-json-string)), or to a temporary string buffer inside
|
||||||
our parser class instances that is valid until the parser object is destroyed or you use it to parse another document.
|
our parser class instances that is valid until the parser object is destroyed or you use it to parse another document.
|
||||||
When using `std::string_view` instances, it is your responsibility to ensure that
|
When using `std::string_view` instances, it is your responsibility to ensure that
|
||||||
`std::string_view` instance does not outlive the pointed-to memory (e.g., either the input
|
`std::string_view` instance does not outlive the pointed-to memory (e.g., either the input
|
||||||
@@ -289,23 +296,31 @@ Some users prefer to use non-JSON native encoding formats such as UTF-16 or UTF-
|
|||||||
transcode the UTF-8 strings produced by the simdjson library to other formats. See the
|
transcode the UTF-8 strings produced by the simdjson library to other formats. See the
|
||||||
[simdutf library](https://github.com/simdutf/simdutf), for example.
|
[simdutf library](https://github.com/simdutf/simdutf), for example.
|
||||||
|
|
||||||
Using the Parsed JSON
|
Avoiding pitfalls: enable development checks
|
||||||
---------------------
|
--------------------
|
||||||
|
|
||||||
We recommend that you first compile and run your code in Debug mode: under Visual Studio,
|
We recommend that you first compile and run your code in Debug mode:
|
||||||
it means having the `_DEBUG` macro defined, and, for other compilers, it means leaving
|
|
||||||
the `__OPTIMIZE__` macro undefined. The simdjson code will set `SIMDJSON_DEVELOPMENT_CHECKS=1`.
|
- under Visual Studio, it means having the `_DEBUG` macro defined,
|
||||||
|
- for other compilers, it means leaving the `__OPTIMIZE__` macro undefined.
|
||||||
|
|
||||||
|
The simdjson code will set `SIMDJSON_DEVELOPMENT_CHECKS=1` in debug mode.
|
||||||
Alternatively, you can set the macro `SIMDJSON_DEVELOPMENT_CHECKS` to 1 prior to including
|
Alternatively, you can set the macro `SIMDJSON_DEVELOPMENT_CHECKS` to 1 prior to including
|
||||||
the `simdjson.h` header to enable these additional checks: just make sure you remove the
|
the `simdjson.h` header to enable these additional checks: just make sure you remove the
|
||||||
definition once your code has been tested. When `SIMDJSON_DEVELOPMENT_CHECKS` is set to 1, the
|
definition once your code has been tested. When `SIMDJSON_DEVELOPMENT_CHECKS` is set to 1, the
|
||||||
simdjson library runs additional (expensive) tests on your code to help ensure that you are
|
simdjson library runs additional (expensive) tests on your code to help ensure that you are
|
||||||
using the library in a safe manner. Once your code has been tested, you can then run it in
|
using the library in a safe manner.
|
||||||
|
|
||||||
|
Once your code has been tested, you can then run it in
|
||||||
Release mode: under Visual Studio, it means having the `_DEBUG` macro undefined, and, for other
|
Release mode: under Visual Studio, it means having the `_DEBUG` macro undefined, and, for other
|
||||||
compilers, it means setting `__OPTIMIZE__` to a positive integer. You can also forcefully
|
compilers, it means setting `__OPTIMIZE__` to a positive integer. You can also forcefully
|
||||||
disable these checks by setting `SIMDJSON_DEVELOPMENT_CHECKS` to 0. Once your code is tested, we
|
disable these checks by setting `SIMDJSON_DEVELOPMENT_CHECKS` to 0. Once your code is tested, we
|
||||||
further encourage you to define `NDEBUG` in your Release builds to disable additional runtime
|
further encourage you to define `NDEBUG` in your Release builds to disable additional runtime
|
||||||
testing and get the best performance.
|
testing and get the best performance.
|
||||||
|
|
||||||
|
Using the parsed JSON
|
||||||
|
---------------------
|
||||||
|
|
||||||
Once you have a document (`simdjson::ondemand::document`), you can navigate it with
|
Once you have a document (`simdjson::ondemand::document`), you can navigate it with
|
||||||
idiomatic C++ iterators, operators and casts. Besides the document instances and
|
idiomatic C++ iterators, operators and casts. Besides the document instances and
|
||||||
native types (`double`, `uint64_t`, `int64_t`, `bool`), we also access
|
native types (`double`, `uint64_t`, `int64_t`, `bool`), we also access
|
||||||
@@ -329,7 +344,7 @@ floating-point values followed by an integer.
|
|||||||
|
|
||||||
We invite you to keep the following rules in mind:
|
We invite you to keep the following rules in mind:
|
||||||
1. While you are accessing the document, the `document` instance should remain in scope: it is your "iterator" which keeps track of where you are in the JSON document. By design, there is one and only one `document` instance per JSON document.
|
1. While you are accessing the document, the `document` instance should remain in scope: it is your "iterator" which keeps track of where you are in the JSON document. By design, there is one and only one `document` instance per JSON document.
|
||||||
2. Because On Demand is really just an iterator, you must fully consume the current object or array before accessing a sibling object or array.
|
2. Because On-Demand is really just an iterator, you must fully consume the current object or array before accessing a sibling object or array.
|
||||||
3. Values can only be consumed once, you should get the values and store them if you plan to need them multiple times. You are expected to access the keys of an object just once. You are expected to go through the values of an array just once.
|
3. Values can only be consumed once, you should get the values and store them if you plan to need them multiple times. You are expected to access the keys of an object just once. You are expected to go through the values of an array just once.
|
||||||
|
|
||||||
The simdjson library makes generous use of `std::string_view` instances. If you are unfamiliar
|
The simdjson library makes generous use of `std::string_view` instances. If you are unfamiliar
|
||||||
@@ -342,14 +357,14 @@ support for users who avoid exceptions. See [the simdjson error handling documen
|
|||||||
|
|
||||||
* **Validate What You Use:** When calling `iterate`, the document is quickly indexed. If it is
|
* **Validate What You Use:** When calling `iterate`, the document is quickly indexed. If it is
|
||||||
not a valid Unicode (UTF-8) string or if there is an unclosed string, an error may be reported right away.
|
not a valid Unicode (UTF-8) string or if there is an unclosed string, an error may be reported right away.
|
||||||
However, it is not fully validated. On Demand only fully validates the values you use and the
|
However, it is not fully validated. On-Demand only fully validates the values you use and the
|
||||||
structure leading to it. It means that at every step as you traverse the document, you may encounter an error. You can handle errors either with exceptions or with error codes.
|
structure leading to it. It means that at every step as you traverse the document, you may encounter an error. You can handle errors either with exceptions or with error codes.
|
||||||
* **Extracting Values:** You can cast a JSON element to a native type:
|
* **Extracting Values:** You can cast a JSON element to a native type:
|
||||||
`double(element)`. This works for `std::string_view`, double, uint64_t, int64_t, bool,
|
`double(element)`. This works for `std::string_view`, double, uint64_t, int64_t, bool,
|
||||||
ondemand::object and ondemand::array. We also have explicit methods such as `get_string()`, `get_double()`,
|
ondemand::object and ondemand::array. We also have explicit methods such as `get_string()`, `get_double()`,
|
||||||
`get_uint64()`, `get_int64()`, `get_bool()`, `get_object()` and `get_array()`. After a cast or an explicit method,
|
`get_uint64()`, `get_int64()`, `get_bool()`, `get_object()` and `get_array()`. After a cast or an explicit method,
|
||||||
the number, string or boolean will be parsed, or the initial `{` or `[` will be verified for `ondemand::object` and `ondemand::array`. An exception may be thrown if
|
the number, string or boolean will be parsed, or the initial `{` or `[` will be verified for `ondemand::object` and `ondemand::array`. An exception may be thrown if
|
||||||
the cast is not possible: there error code is `simdjson::INCORRECT_TYPE` (see [Error Handling](#error-handling)). Importantly, when getting an ondemand::object or ondemand::array instance, its content is
|
the cast is not possible: there error code is `simdjson::INCORRECT_TYPE` (see [Error handling](#error-handling)). Importantly, when getting an ondemand::object or ondemand::array instance, its content is
|
||||||
not validated: you are only guaranteed that the corresponding initial character (`{` or `[`) is present. Thus,
|
not validated: you are only guaranteed that the corresponding initial character (`{` or `[`) is present. Thus,
|
||||||
for example, you could have an ondemand::object instance pointing at the invalid JSON `{ "this is not a valid object" }`: the validation occurs as you access the content.
|
for example, you could have an ondemand::object instance pointing at the invalid JSON `{ "this is not a valid object" }`: the validation occurs as you access the content.
|
||||||
The `get_string()` returns a valid UTF-8 string, after
|
The `get_string()` returns a valid UTF-8 string, after
|
||||||
@@ -365,17 +380,40 @@ support for users who avoid exceptions. See [the simdjson error handling documen
|
|||||||
> IMPORTANT NOTE: values can only be parsed once. Since documents are *iterators*, once you have
|
> IMPORTANT NOTE: values can only be parsed once. Since documents are *iterators*, once you have
|
||||||
> parsed a value (such as by casting to double), you cannot get at it again. It is an error to call
|
> parsed a value (such as by casting to double), you cannot get at it again. It is an error to call
|
||||||
> `get_string()` twice on an object (or to cast an object twice to `std::string_view`).
|
> `get_string()` twice on an object (or to cast an object twice to `std::string_view`).
|
||||||
|
* **Array Iteration:** To iterate through an array, use `for (auto value : array) { ... }`. This will
|
||||||
|
step through each value in the JSON array.
|
||||||
|
|
||||||
|
To iterate through an array, you should be at the beginning
|
||||||
|
of the array: to warn you, an OUT_OF_ORDER_ITERATION error is generated [when development checks](#avoiding-pitfalls-enable-development-checks) are active. If you need to access an array more
|
||||||
|
than once, you may call `reset()` on it although we discourage this practice. Keep in mind that
|
||||||
|
you should consume each value at most once.
|
||||||
|
|
||||||
|
If you know the type of the value, you can cast it right there, too! `for (double value : array) { ... }`.
|
||||||
|
|
||||||
|
You may also use explicit iterators: `for(auto i = array.begin(); i != array.end(); i++) {}`. You can check that an array is empty with the condition `auto i = array.begin(); if (i == array.end()) {...}`.
|
||||||
|
* **Object Iteration:** You can iterate through an object's fields, as well: `for (auto field : object) { ... }`. You may also use explicit iterators : `for(auto i = object.begin(); i != object.end(); i++) { auto field = *i; .... }`. You can check that an object is empty with the condition `auto i = object.begin(); if (i == object.end()) {...}`.
|
||||||
|
- `field.unescaped_key()` will get you the unescaped key string as a `std::string_view` instance. E.g., the JSON string `"\u00e1"` becomes the Unicode string `á`. Optionally, you pass `true` as a parameter to the `unescaped_key` method if you want invalid escape sequences to be replaced by a default replacement character (e.g., `\ud800\ud801\ud811`): otherwise bad escape sequences lead to an immediate error.
|
||||||
|
- `field.escaped_key()` will get you the key string as as a `std::string_view` instance, but unlike `unescaped_key()`, the key is not processed, so no unescaping is done. E.g., the JSON string `"\u00e1"` becomes the Unicode string `\u00e1`. We expect that `escaped_key()` is faster than `field.unescaped_key()`.
|
||||||
|
- `field.value()` will get you the value, which you can then use all these other methods on.
|
||||||
|
|
||||||
|
|
||||||
|
To iterate through an object, you should be at the beginning
|
||||||
|
of the object: to warn you, an OUT_OF_ORDER_ITERATION error is generated [when development checks](#avoiding-pitfalls-enable-development-checks) are active. If you need to access an object more
|
||||||
|
than once, you may call `reset()` on it although we discourage this practice. Keep in mind that
|
||||||
|
you should consume each value at most once.
|
||||||
|
* **Array Index:** Because it is forward-only, you cannot look up an array element by index by index. Instead,
|
||||||
|
you should iterate through the array and keep an index yourself. Exceptionally, if need a single value
|
||||||
|
out of the array, you may use an array access (e.g., `array[1]`).
|
||||||
* **Field Access:** To get the value of the "foo" field in an object, use `object["foo"]`. This will
|
* **Field Access:** To get the value of the "foo" field in an object, use `object["foo"]`. This will
|
||||||
scan through the object looking for the field with the matching string, doing a character-by-character
|
scan through the object looking for the field with the matching string, doing a character-by-character
|
||||||
comparison. It may generate the error `simdjson::NO_SUCH_FIELD` if there is no such key in the object, it may throw an exception (see [Error Handling](#error-handling)). For efficiency reason, you should avoid looking up the same field repeatedly: e.g., do
|
comparison. It may generate the error `simdjson::NO_SUCH_FIELD` if there is no such key in the object, it may throw an exception (see [Error handling](#error-handling)). For efficiency reason, you should avoid looking up the same field repeatedly: e.g., do
|
||||||
not do `object["foo"]` followed by `object["foo"]` with the same `object` instance. For best performance, you should try to query the keys in the same order they appear in the document. If you need several keys and you cannot predict the order they will appear in, it is recommended to iterate through all keys `for(auto field : object) {...}`. Keep in mind that On Demand does not buffer or save the result of the parsing: if you repeatedly access `object["foo"]`, then it must repeatedly seek the key and parse the content. The library does not provide a distinct function to check if a key is present, instead we recommend you attempt to access the key: e.g., by doing `ondemand::value val{}; if (!object["foo"].get(val)) {...}`, you have that `val` contains the requested value inside the if clause. It is your responsibility as a user to temporarily keep a reference to the value (`auto v = object["foo"]`), or to consume the content and store it in your own data structures. If you consume an
|
not do `object["foo"]` followed by `object["foo"]` with the same `object` instance. For best performance, you should try to query the keys in the same order they appear in the document. If you need several keys and you cannot predict the order they will appear in, it is recommended to iterate through all keys `for(auto field : object) {...}`. Generally, you should not mix and match iterating through an object (`for(auto field : object) {...}`) and key accesses (`object["foo"]`): if you need to iterate through an object after a key access, you need to call `reset()` on the object. Whenever you call `reset()`, you need to keep in mind that though you can iterate over the array repeatedly, values should be consumedonly once (e.g., repeatedly calling `unescaped_key()` on the same key is forbidden). Keep in mind that On-Demand does not buffer or save the result of the parsing: if you repeatedly access `object["foo"]`, then it must repeatedly seek the key and parse the content. The library does not provide a distinct function to check if a key is present, instead we recommend you attempt to access the key: e.g., by doing `ondemand::value val{}; if (!object["foo"].get(val)) {...}`, you have that `val` contains the requested value inside the if clause. It is your responsibility as a user to temporarily keep a reference to the value (`auto v = object["foo"]`), or to consume the content and store it in your own data structures. If you consume an
|
||||||
object twice: `std::string_view(object["foo"]` followed by `std::string_view(object["foo"]` then your code
|
object twice: `std::string_view(object["foo"]` followed by `std::string_view(object["foo"]` then your code
|
||||||
is in error. Furthermore, you can only consume one field at a time, on the same object. The
|
is in error. Furthermore, you can only consume one field at a time, on the same object. The
|
||||||
value instance you get from `content["bids"]` becomes invalid when you call `content["asks"]`.
|
value instance you get from `content["bids"]` becomes invalid when you call `content["asks"]`.
|
||||||
If you have retrieved `content["bids"].get_array()` and you later call
|
If you have retrieved `content["bids"].get_array()` and you later call
|
||||||
`content["asks"].get_array()`, then the first array should no longer be accessed: it would be
|
`content["asks"].get_array()`, then the first array should no longer be accessed: it would be
|
||||||
unsafe to do so. You can detect such mistakes by first compiling and running the code in
|
unsafe to do so. You can detect such mistakes by first compiling and running the code [with development checks](#avoiding-pitfalls-enable-development-checks): an OUT_OF_ORDER_ITERATION error is generated.
|
||||||
Debug mode: an OUT_OF_ORDER_ITERATION error is generated.
|
|
||||||
|
|
||||||
> NOTE: JSON allows you to escape characters in keys. E.g., the key `"date"` may be written as
|
> NOTE: JSON allows you to escape characters in keys. E.g., the key `"date"` may be written as
|
||||||
> `"\u0064\u0061\u0074\u0065"`. By default, simdjson does *not* unescape keys when matching.
|
> `"\u0064\u0061\u0074\u0065"`. By default, simdjson does *not* unescape keys when matching.
|
||||||
@@ -432,18 +470,6 @@ support for users who avoid exceptions. See [the simdjson error handling documen
|
|||||||
> double y = doc["y"]; // The cursor is now after the 2 (at })
|
> double y = doc["y"]; // The cursor is now after the 2 (at })
|
||||||
> double x = doc["x"]; // Success: [] loops back around to find "x"
|
> double x = doc["x"]; // Success: [] loops back around to find "x"
|
||||||
> ```
|
> ```
|
||||||
* **Array Iteration:** To iterate through an array, use `for (auto value : array) { ... }`. This will
|
|
||||||
step through each value in the JSON array.
|
|
||||||
|
|
||||||
If you know the type of the value, you can cast it right there, too! `for (double value : array) { ... }`.
|
|
||||||
|
|
||||||
You may also use explicit iterators: `for(auto i = array.begin(); i != array.end(); i++) {}`. You can check that an array is empty with the condition `auto i = array.begin(); if (i == array.end()) {...}`.
|
|
||||||
* **Object Iteration:** You can iterate through an object's fields, as well: `for (auto field : object) { ... }`. You may also use explicit iterators : `for(auto i = object.begin(); i != object.end(); i++) { auto field = *i; .... }`. You can check that an object is empty with the condition `auto i = object.begin(); if (i == object.end()) {...}`.
|
|
||||||
- `field.unescaped_key()` will get you the unescaped key string as a `std::string_view` instance. E.g., the JSON string `"\u00e1"` becomes the Unicode string `á`. Optionally, you pass `true` as a parameter to the `unescaped_key` method if you want invalid escape sequences to be replaced by a default replacement character (e.g., `\ud800\ud801\ud811`): otherwise bad escape sequences lead to an immediate error.
|
|
||||||
- `field.escaped_key()` will get you the key string as as a `std::string_view` instance, but unlike `unescaped_key()`, the key is not processed, so no unescaping is done. E.g., the JSON string `"\u00e1"` becomes the Unicode string `\u00e1`. We expect that `escaped_key()` is faster than `field.unescaped_key()`.
|
|
||||||
- `field.value()` will get you the value, which you can then use all these other methods on.
|
|
||||||
* **Array Index:** Because it is forward-only, you cannot look up an array element by index by index. Instead,
|
|
||||||
you should iterate through the array and keep an index yourself.
|
|
||||||
* **Output to strings:** Given a document, a value, an array or an object in a JSON document, you can output a JSON string version suitable to be parsed again as JSON content: `simdjson::to_json_string(element)`. A call to `to_json_string` consumes fully the element: if you apply it on a document, the JSON pointer is advanced to the end of the document. The `simdjson::to_json_string` does not allocate memory. The `to_json_string` function should not be confused with retrieving the value of a string instance which are escaped and represented using a lightweight `std::string_view` instance pointing at an internal string buffer inside the parser instance. To illustrate, the first of the following two code segments will print the unescaped string `"test"` complete with the quote whereas the second one will print the escaped content of the string (without the quotes).
|
* **Output to strings:** Given a document, a value, an array or an object in a JSON document, you can output a JSON string version suitable to be parsed again as JSON content: `simdjson::to_json_string(element)`. A call to `to_json_string` consumes fully the element: if you apply it on a document, the JSON pointer is advanced to the end of the document. The `simdjson::to_json_string` does not allocate memory. The `to_json_string` function should not be confused with retrieving the value of a string instance which are escaped and represented using a lightweight `std::string_view` instance pointing at an internal string buffer inside the parser instance. To illustrate, the first of the following two code segments will print the unescaped string `"test"` complete with the quote whereas the second one will print the escaped content of the string (without the quotes).
|
||||||
> ```C++
|
> ```C++
|
||||||
> // serialize a JSON to an escaped std::string instance so that it can be parsed again as JSON
|
> // serialize a JSON to an escaped std::string instance so that it can be parsed again as JSON
|
||||||
@@ -545,7 +571,7 @@ support for users who avoid exceptions. See [the simdjson error handling documen
|
|||||||
auto json = R"( { "test":{ "val1":1, "val2":2 } } )"_padded;
|
auto json = R"( { "test":{ "val1":1, "val2":2 } } )"_padded;
|
||||||
auto doc = parser.iterate(json);
|
auto doc = parser.iterate(json);
|
||||||
size_t count = doc.count_fields(); // requires simdjson 1.0 or better
|
size_t count = doc.count_fields(); // requires simdjson 1.0 or better
|
||||||
std::cout << "Number of fields: " << new_count << std::endl; // Prints "Number of fields: 1"
|
std::cout << "Number of fields: " << count << std::endl; // Prints "Number of fields: 1"
|
||||||
```
|
```
|
||||||
Similarly to `count_elements`, you should not let an object instance go out of scope before consuming it after calling
|
Similarly to `count_elements`, you should not let an object instance go out of scope before consuming it after calling
|
||||||
the `count_fields` method. If you access an object inside a document, you can use the `count_fields` method as follow.
|
the `count_fields` method. If you access an object inside a document, you can use the `count_fields` method as follow.
|
||||||
@@ -643,7 +669,7 @@ support for users who avoid exceptions. See [the simdjson error handling documen
|
|||||||
}
|
}
|
||||||
```
|
```
|
||||||
|
|
||||||
### Using the Parsed JSON: Additional examples
|
### Using the parsed JSON: additional examples
|
||||||
|
|
||||||
|
|
||||||
Let us review these concepts with some additional examples. For simplicity, we omit the include clauses (`#include "simdjson.h"`) as well as namespace-using clauses (`using namespace simdjson;`).
|
Let us review these concepts with some additional examples. For simplicity, we omit the include clauses (`#include "simdjson.h"`) as well as namespace-using clauses (`using namespace simdjson;`).
|
||||||
@@ -815,7 +841,7 @@ simdjson::ondemand::value::get() noexcept {
|
|||||||
}
|
}
|
||||||
```
|
```
|
||||||
|
|
||||||
We may then provide support for our `Car`` struct:
|
We may then provide support for our `Car` struct:
|
||||||
|
|
||||||
```C++
|
```C++
|
||||||
template <>
|
template <>
|
||||||
@@ -1081,9 +1107,9 @@ If you find yourself needing only fast Unicode functions, consider using the sim
|
|||||||
JSON Pointer
|
JSON Pointer
|
||||||
------------
|
------------
|
||||||
|
|
||||||
The simdjson library also supports [JSON pointer](https://tools.ietf.org/html/rfc6901) through the `at_pointer()` method, letting you reach further down into the document in a single call. JSON pointer is supported by both the [DOM approach](https://github.com/simdjson/simdjson/blob/master/doc/dom.md#json-pointer) as well as the On Demand approach.
|
The simdjson library also supports [JSON pointer](https://tools.ietf.org/html/rfc6901) through the `at_pointer()` method, letting you reach further down into the document in a single call. JSON pointer is supported by both the [DOM approach](https://github.com/simdjson/simdjson/blob/master/doc/dom.md#json-pointer) as well as the On-Demand approach.
|
||||||
|
|
||||||
**Note:** The On Demand implementation of JSON pointer relies on `find_field` which implies that it does not unescape keys when matching.
|
**Note:** The On-Demand implementation of JSON pointer relies on `find_field` which implies that it does not unescape keys when matching.
|
||||||
|
|
||||||
Consider the following example:
|
Consider the following example:
|
||||||
|
|
||||||
@@ -1198,9 +1224,11 @@ be represented as `value` instances. You can check that a document is a scalar w
|
|||||||
JSONPath
|
JSONPath
|
||||||
------------
|
------------
|
||||||
|
|
||||||
The simdjson library now supports a subset of [JSONPath](https://datatracker.ietf.org/doc/html/draft-normington-jsonpath-00) through the `at_path()` method, allowing you to reach further into the document in a single call. The subset of JSONPath that is implemented is the subset that is trivially convertible into the JSON Pointer format, using `.` to access a field and `[]` to access a specific index.
|
The simdjson library supports a subset of [JSONPath](https://datatracker.ietf.org/doc/html/draft-normington-jsonpath-00) through the `at_path()` method, allowing you to reach further into the document in a single call. The subset of JSONPath that is implemented is the subset that is trivially convertible into the JSON Pointer format, using `.` to access a field and `[]` to access a specific index.
|
||||||
|
|
||||||
This implementation relies on `at_path()` converting its argument to JSON Pointer and then calling `at_pointer`, which makes use of [`rewind`](#rewind) to reset the parser at the beginning of the document. Hence, it invalidates all previously parsed values, objects and arrays: make sure to consume the values between each call to `at_path`.
|
This implementation relies on `at_path()` converting its argument to JSON Pointer and then calling `at_pointer`, which makes use of
|
||||||
|
[`rewind`](#rewind) to reset the parser at the beginning of the document. Hence, it invalidates all previously parsed values, objects
|
||||||
|
and arrays: make sure to consume the values between each call to `at_path`.
|
||||||
|
|
||||||
Consider the following example:
|
Consider the following example:
|
||||||
|
|
||||||
@@ -1241,7 +1269,19 @@ doc.at_path(".\\u00E9") == 123; // true
|
|||||||
doc.at_path((const char*)u8".\u00E9") // returns an error (NO_SUCH_FIELD)
|
doc.at_path((const char*)u8".\u00E9") // returns an error (NO_SUCH_FIELD)
|
||||||
```
|
```
|
||||||
|
|
||||||
Error Handling
|
|
||||||
|
We also support the `$` prefix. When you start a JSONPath expression with $, you are indicating that the path starts from the root of the JSON document. E.g.,
|
||||||
|
|
||||||
|
```c++
|
||||||
|
auto json = R"( { "c" :{ "foo": { "a": [ 10, 20, 30 ] }}, "d": { "foo2": { "a": [ 10, 20, 30 ] }} , "e": 120 })"_padded;
|
||||||
|
ondemand::parser parser;
|
||||||
|
ondemand::document doc = parser.iterate(json);
|
||||||
|
ondemand::object obj = doc.get_object();
|
||||||
|
int64_t x = obj.at_path("$.c.foo.a[1]"); // 20
|
||||||
|
x = obj.at_path("$.d.foo2.a.2"); // 30
|
||||||
|
```
|
||||||
|
|
||||||
|
Error handling
|
||||||
--------------
|
--------------
|
||||||
|
|
||||||
Error handing with exception and a single try/catch clause makes the code simple, but it gives you little control over errors. For easier debugging or more robust error handling, you may want to consider our exception-free approach.
|
Error handing with exception and a single try/catch clause makes the code simple, but it gives you little control over errors. For easier debugging or more robust error handling, you may want to consider our exception-free approach.
|
||||||
@@ -1391,9 +1431,9 @@ int main(void) {
|
|||||||
The `at` method can only be called once on an array. It cannot be used
|
The `at` method can only be called once on an array. It cannot be used
|
||||||
to iterate through the values of an array.
|
to iterate through the values of an array.
|
||||||
|
|
||||||
### Error Handling Examples without Exceptions
|
### Error handling examples without exceptions
|
||||||
|
|
||||||
This is how the example in "Using the Parsed JSON" could be written using only error code checking (without exceptions):
|
This is how the example in "Using the parsed JSON" could be written using only error code checking (without exceptions):
|
||||||
|
|
||||||
```c++
|
```c++
|
||||||
bool parse() {
|
bool parse() {
|
||||||
@@ -1489,7 +1529,7 @@ having to handle exceptions.
|
|||||||
}
|
}
|
||||||
```
|
```
|
||||||
|
|
||||||
### Disabling Exceptions
|
### Disabling exceptions
|
||||||
|
|
||||||
The simdjson can be build with exceptions entirely disabled. It checks the `__cpp_exceptions` macro at compile time. Even if exceptions are enabled in your compiler, you may still disable exceptions specifically for simdjson, by setting `SIMDJSON_EXCEPTIONS` to `0` (false) at compile-time when building the simdjson library. If you are building with CMake, to ensure you don't write any code that uses exceptions, you compile with `SIMDJSON_EXCEPTIONS=OFF`. For example, if including the project via cmake:
|
The simdjson can be build with exceptions entirely disabled. It checks the `__cpp_exceptions` macro at compile time. Even if exceptions are enabled in your compiler, you may still disable exceptions specifically for simdjson, by setting `SIMDJSON_EXCEPTIONS` to `0` (false) at compile-time when building the simdjson library. If you are building with CMake, to ensure you don't write any code that uses exceptions, you compile with `SIMDJSON_EXCEPTIONS=OFF`. For example, if including the project via cmake:
|
||||||
|
|
||||||
@@ -1701,7 +1741,7 @@ before printout the data.
|
|||||||
}
|
}
|
||||||
```
|
```
|
||||||
|
|
||||||
Performance note: the On Demand front-end does not materialize the parsed numbers and other values. If you are accessing everything twice, you may need to parse them twice. Thus the rewind functionality is best suited for cases where the first pass only scans the structure of the document.
|
Performance note: the On-Demand front-end does not materialize the parsed numbers and other values. If you are accessing everything twice, you may need to parse them twice. Thus the rewind functionality is best suited for cases where the first pass only scans the structure of the document.
|
||||||
|
|
||||||
Both arrays and objects have a similar method `reset()`. It is similar
|
Both arrays and objects have a similar method `reset()`. It is similar
|
||||||
to the document `rewind()` method, except that it does not rewind the
|
to the document `rewind()` method, except that it does not rewind the
|
||||||
@@ -1740,7 +1780,7 @@ for (auto doc : docs) {
|
|||||||
```
|
```
|
||||||
|
|
||||||
|
|
||||||
Unlike `parser.iterate`, `parser.iterate_many` may parse "on demand" (lazily). That is, no parsing may have been done before you enter the loop
|
Unlike `parser.iterate`, `parser.iterate_many` may parse "On-Demand" (lazily). That is, no parsing may have been done before you enter the loop
|
||||||
`for (auto doc : docs) {` and you should expect the parser to only ever fully parse one JSON document at a time.
|
`for (auto doc : docs) {` and you should expect the parser to only ever fully parse one JSON document at a time.
|
||||||
|
|
||||||
As with `parser.iterate`, when calling `parser.iterate_many(string)`, no copy is made of the provided string input. The provided memory buffer may be accessed each time a JSON document is parsed. Calling `parser.iterate_many(string)` on a temporary string buffer (e.g., `docs = parser.parse_many("[1,2,3]"_padded)`) is unsafe (and will not compile) because the `document_stream` instance needs access to the buffer to return the JSON documents.
|
As with `parser.iterate`, when calling `parser.iterate_many(string)`, no copy is made of the provided string input. The provided memory buffer may be accessed each time a JSON document is parsed. Calling `parser.iterate_many(string)` on a temporary string buffer (e.g., `docs = parser.parse_many("[1,2,3]"_padded)`) is unsafe (and will not compile) because the `document_stream` instance needs access to the buffer to return the JSON documents.
|
||||||
@@ -1800,7 +1840,7 @@ If your documents are large (e.g., larger than a megabyte), then the `iterate_ma
|
|||||||
We also provide some support for comma-separated documents and other advanced features.
|
We also provide some support for comma-separated documents and other advanced features.
|
||||||
See [iterate_many.md](iterate_many.md) for detailed information and design.
|
See [iterate_many.md](iterate_many.md) for detailed information and design.
|
||||||
|
|
||||||
Parsing Numbers Inside Strings
|
Parsing numbers inside strings
|
||||||
------------------------------
|
------------------------------
|
||||||
|
|
||||||
Though the JSON specification allows for numbers and string values, many engineers choose to integrate the numbers inside strings, e.g., they prefer `{"a":"1.9"}` to`{"a":1.9}`.
|
Though the JSON specification allows for numbers and string values, many engineers choose to integrate the numbers inside strings, e.g., they prefer `{"a":"1.9"}` to`{"a":1.9}`.
|
||||||
@@ -2030,7 +2070,7 @@ This code prints the following:
|
|||||||
'99999999999999999999999 '
|
'99999999999999999999999 '
|
||||||
```
|
```
|
||||||
|
|
||||||
Raw Strings From Keys
|
Raw strings from keys
|
||||||
-----------
|
-----------
|
||||||
|
|
||||||
It is sometimes useful to have access to a raw (unescaped) string: we make available a
|
It is sometimes useful to have access to a raw (unescaped) string: we make available a
|
||||||
@@ -2082,7 +2122,7 @@ begins with `"name"` and may containing trailing white-space characters.
|
|||||||
```
|
```
|
||||||
|
|
||||||
|
|
||||||
General Direct Access to the Raw JSON String
|
General direct access to the raw JSON string
|
||||||
--------------------------------
|
--------------------------------
|
||||||
If your value is a string, the `raw_json_string` you get with `get_raw_json_string()` gives you direct access to the unprocessed
|
If your value is a string, the `raw_json_string` you get with `get_raw_json_string()` gives you direct access to the unprocessed
|
||||||
string. But the simdjson library allows you to have access to the raw underlying JSON
|
string. But the simdjson library allows you to have access to the raw underlying JSON
|
||||||
@@ -2189,12 +2229,39 @@ string representation.
|
|||||||
```
|
```
|
||||||
|
|
||||||
|
|
||||||
Storing Directly into an Existing String Instance
|
You can use `raw_json()` to capture the content of some JSON values as `std::string_view`
|
||||||
|
instances which can be safely used later. The `std::string_view` instances point inside
|
||||||
|
the original document and do not depend in any way on simdjson. In the following example,
|
||||||
|
we store the `std::string_view` instances inside a `std::vector<std::string_view>` instance
|
||||||
|
and print the out after the parsing is concluded:
|
||||||
|
|
||||||
|
```cpp
|
||||||
|
padded_string json_padded = "{\"a\":[1,2,3], \"b\": 2, \"c\": \"hello\"}"_padded;
|
||||||
|
std::vector<std::string_view> fields;
|
||||||
|
|
||||||
|
ondemand::parser parser;
|
||||||
|
auto doc = parser.iterate(json_padded);
|
||||||
|
auto object = doc.get_object();
|
||||||
|
for (auto field : object) {
|
||||||
|
fields.push_back(field.value().raw_json());
|
||||||
|
}
|
||||||
|
// Output the fields
|
||||||
|
// Expected output:
|
||||||
|
// [1,2,3]
|
||||||
|
// 2
|
||||||
|
// "hello"
|
||||||
|
for (std::string_view field_ref : fields) {
|
||||||
|
std::cout << field_ref << std::endl;
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
|
||||||
|
Storing directly into an existing string instance
|
||||||
-----------------------------------------------------
|
-----------------------------------------------------
|
||||||
|
|
||||||
The simdjson library favours the use of `std::string_view` instances because
|
The simdjson library favours the use of `std::string_view` instances because
|
||||||
it tends to lead to better performance due to causing fewer memory allocations.
|
it tends to lead to better performance due to causing fewer memory allocations.
|
||||||
However, they are cases where you need to store a string result in a `std::string``
|
However, they are cases where you need to store a string result in a `std::string`
|
||||||
instance. You can do so with a templated version of the `to_string()` method which takes as
|
instance. You can do so with a templated version of the `to_string()` method which takes as
|
||||||
a parameter a reference to a `std::string`.
|
a parameter a reference to a `std::string`.
|
||||||
|
|
||||||
@@ -2236,7 +2303,7 @@ can use it with features such as `std::optional`:
|
|||||||
You should be mindful of the trade-off: allocating multiple
|
You should be mindful of the trade-off: allocating multiple
|
||||||
`std::string` instances can become expensive.
|
`std::string` instances can become expensive.
|
||||||
|
|
||||||
Thread Safety
|
Thread safety
|
||||||
-------------
|
-------------
|
||||||
|
|
||||||
We built simdjson with thread safety in mind.
|
We built simdjson with thread safety in mind.
|
||||||
@@ -2244,15 +2311,20 @@ We built simdjson with thread safety in mind.
|
|||||||
The simdjson library is single-threaded except for [`iterate_many`](iterate_many.md) and [`parse_many`](parse_many.md) which may use secondary threads under their control when the library is compiled with thread support.
|
The simdjson library is single-threaded except for [`iterate_many`](iterate_many.md) and [`parse_many`](parse_many.md) which may use secondary threads under their control when the library is compiled with thread support.
|
||||||
|
|
||||||
|
|
||||||
We recommend using one `parser` object per thread. When using the On Demand front-end (our default), you should access the `document` instances in a single-threaded manner since it
|
We recommend using one `parser` object per thread. When using the On-Demand front-end (our default), you should access the `document` instances in a single-threaded manner since it
|
||||||
acts as an iterator (and is therefore not thread safe).
|
acts as an iterator (and is therefore not thread safe).
|
||||||
|
|
||||||
The CPU detection, which runs the first time parsing is attempted and switches to the fastest
|
The CPU detection, which runs the first time parsing is attempted and switches to the fastest
|
||||||
parser for your CPU, is transparent and thread-safe.
|
parser for your CPU, is transparent and thread-safe.
|
||||||
|
Our runtime dispatching is based on global objects that are instantiated at the beginning of the
|
||||||
|
main thread and may be discarded at the end of the main thread. If you have multiple threads running
|
||||||
|
and some threads use the library while the main thread is cleaning up ressources, you may encounter
|
||||||
|
issues. If you expect such problems, you may consider using [std::quick_exit](https://en.cppreference.com/w/cpp/utility/program/quick_exit).
|
||||||
|
|
||||||
In a threaded environment, stack space is often limited. Running code like simdjson in debug mode may require hundreds of kilobytes of stack memory. Thus stack overflows are a possibility. We recommend you turn on optimization when working in an environment where stack space is limited. If you must run your code in debug mode, we recommend you configure your system to have more stack space. We discourage you from running production code based on a debug build.
|
In a threaded environment, stack space is often limited. Running code like simdjson in debug mode may require hundreds of kilobytes of stack memory. Thus stack overflows are a possibility. We recommend you turn on optimization when working in an environment where stack space is limited. If you must run your code in debug mode, we recommend you configure your system to have more stack space. We discourage you from running production code based on a debug build.
|
||||||
|
|
||||||
Standard Compliance
|
|
||||||
|
Standard compliance
|
||||||
--------------------
|
--------------------
|
||||||
|
|
||||||
The simdjson library is fully compliant with the [RFC 8259](https://www.tbray.org/ongoing/When/201x/2017/12/14/rfc8259.html) JSON specification.
|
The simdjson library is fully compliant with the [RFC 8259](https://www.tbray.org/ongoing/When/201x/2017/12/14/rfc8259.html) JSON specification.
|
||||||
@@ -2270,7 +2342,7 @@ The simdjson library is fully compliant with the [RFC 8259](https://www.tbray.o
|
|||||||
- The specification states that an implementation may set limits on the maximum depth of nesting. By default, the simdjson will refuse to parse documents with a depth exceeding 1024.
|
- The specification states that an implementation may set limits on the maximum depth of nesting. By default, the simdjson will refuse to parse documents with a depth exceeding 1024.
|
||||||
|
|
||||||
|
|
||||||
Backwards Compatibility
|
Backwards compatibility
|
||||||
-----------------------
|
-----------------------
|
||||||
|
|
||||||
The only header file supported by simdjson is `simdjson.h`. Older versions of simdjson published a
|
The only header file supported by simdjson is `simdjson.h`. Older versions of simdjson published a
|
||||||
@@ -2565,11 +2637,37 @@ int main(void) {
|
|||||||
}
|
}
|
||||||
```
|
```
|
||||||
|
|
||||||
Performance Tips
|
|
||||||
|
* Example 4: Value capture with `std::string_view` instances
|
||||||
|
|
||||||
|
```cpp
|
||||||
|
void example() {
|
||||||
|
ondemand::parser parser;
|
||||||
|
const padded_string json = R"({ "parent": {"child1": {"name": "John"} , "child2": {"name": "Daniel"}} })"_padded;
|
||||||
|
auto doc = parser.iterate(json);
|
||||||
|
ondemand::object parent = doc["parent"];
|
||||||
|
// parent owns the focus
|
||||||
|
ondemand::object c1 = parent["child1"];
|
||||||
|
// c1 owns the focus
|
||||||
|
//
|
||||||
|
std::string_view as1 = c1["name"];
|
||||||
|
// We have that as1 == "John", as long as 'parser' and 'json' live
|
||||||
|
// c2 attempts to grab the focus from parent but fails
|
||||||
|
ondemand::object c2 = parent["child2"];
|
||||||
|
// c2 owns the focus, at this point c1 is invalid
|
||||||
|
std::string_view as2 = c2["name"];
|
||||||
|
// We have that as2 == "Daniel", as long as 'parser' and 'json' live
|
||||||
|
std::cout << as1 << " " << as2 << std::endl; // prints John Daniel
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
Performance tips
|
||||||
--------
|
--------
|
||||||
|
|
||||||
|
|
||||||
- The On Demand front-end works best when doing a single pass over the input: avoid calling `count_elements`, `rewind` and similar methods.
|
- Read [our performance notes](performance.md) for advanced topics.
|
||||||
|
- To better understand the operation of your On-Demand parser, and whether it is performing as well as you think it should be, there is a logger feature built in to simdjson! To use it, define the pre-processor directive `SIMDJSON_VERBOSE_LOGGING` prior to including the `simdjson.h` header, which enables logging in simdjson. Run your code. It may generate a lot of logging output; adding printouts from your application that show each section may be helpful. The log's output will show step-by-step information on state, buffer pointer position, depth, and key retrieval status. Importantly, unless `SIMDJSON_VERBOSE_LOGGING` is defined, logging is entirely disabled and thus carries no overhead.
|
||||||
|
- The On-Demand front-end works best when doing a single pass over the input: avoid calling `count_elements`, `rewind`, `reset` and similar methods.
|
||||||
- If you are familiar with assembly language, you may use the online tool godbolt to explore the compiled code. The following example may work: [https://godbolt.org/z/xE4GWs573](https://godbolt.org/z/xE4GWs573).
|
- If you are familiar with assembly language, you may use the online tool godbolt to explore the compiled code. The following example may work: [https://godbolt.org/z/xE4GWs573](https://godbolt.org/z/xE4GWs573).
|
||||||
- Given a field `field` in an object, calling `field.key()` is often faster than `field.unescaped_key()` so if you do not need an unescaped `std::string_view` instance, prefer `field.key()`. Similarly, we expect `field.escaped_key()` to be faster than `field.unescaped_key()` even though both return a `std::string_view` instance.
|
- Given a field `field` in an object, calling `field.key()` is often faster than `field.unescaped_key()` so if you do not need an unescaped `std::string_view` instance, prefer `field.key()`. Similarly, we expect `field.escaped_key()` to be faster than `field.unescaped_key()` even though both return a `std::string_view` instance.
|
||||||
- For release builds, we recommend setting `NDEBUG` pre-processor directive when compiling the `simdjson` library. Importantly, using the optimization flags `-O2` or `-O3` under GCC and LLVM clang does not set the `NDEBUG` directive, you must set it manually (e.g., `-DNDEBUG`).
|
- For release builds, we recommend setting `NDEBUG` pre-processor directive when compiling the `simdjson` library. Importantly, using the optimization flags `-O2` or `-O3` under GCC and LLVM clang does not set the `NDEBUG` directive, you must set it manually (e.g., `-DNDEBUG`).
|
||||||
@@ -2589,10 +2687,12 @@ Performance Tips
|
|||||||
std::string_view year = data["year"];
|
std::string_view year = data["year"];
|
||||||
std::string_view rating = data["rating"];
|
std::string_view rating = data["rating"];
|
||||||
```
|
```
|
||||||
- To better understand the operation of your On Demand parser, and whether it is performing as well as you think it should be, there is a logger feature built in to simdjson! To use it, define the pre-processor directive `SIMDJSON_VERBOSE_LOGGING` prior to including the `simdjson.h` header, which enables logging in simdjson. Run your code. It may generate a lot of logging output; adding printouts from your application that show each section may be helpful. The log's output will show step-by-step information on state, buffer pointer position, depth, and key retrieval status. Importantly, unless `SIMDJSON_VERBOSE_LOGGING` is defined, logging is entirely disabled and thus carries no overhead.
|
- You will get better performance if you seek the keys in the order in which they appear in the document. So if processing `{"a":1, "b":2, "c":3}`, do `value1 = data["a"]; value2 = data["b"]; value3 data["c"];` and not `value2 = data["b"]; value1 = data["a"]; value3 data["c"];`. Of course, it is not always possible to know for sure in which order the keys appear.
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Further reading
|
Further reading
|
||||||
--------
|
--------
|
||||||
|
|
||||||
|
|
||||||
- John Keiser, Daniel Lemire, [On-Demand JSON: A Better Way to Parse Documents?](http://arxiv.org/abs/2312.17149), Software: Practice and Experience (to appear)
|
- John Keiser, Daniel Lemire, [On-Demand JSON: A Better Way to Parse Documents?](http://arxiv.org/abs/2312.17149), Software: Practice and Experience 54 (6), 2024
|
||||||
|
|||||||
+46
-2
@@ -3,11 +3,12 @@ The Document-Object-Model (DOM) front-end
|
|||||||
|
|
||||||
An overview of what you need to know to use simdjson, with examples.
|
An overview of what you need to know to use simdjson, with examples.
|
||||||
|
|
||||||
* [DOM vs On Demand](#dom-vs-on-demand)
|
* [DOM vs On-Demand](#dom-vs-on-demand)
|
||||||
* [The Basics: Loading and Parsing JSON Documents](#the-basics-loading-and-parsing-json-documents-using-the-dom-front-end)
|
* [The Basics: Loading and Parsing JSON Documents](#the-basics-loading-and-parsing-json-documents-using-the-dom-front-end)
|
||||||
* [Using the Parsed JSON](#using-the-parsed-json)
|
* [Using the Parsed JSON](#using-the-parsed-json)
|
||||||
* [C++17 Support](#c17-support)
|
* [C++17 Support](#c17-support)
|
||||||
* [JSON Pointer](#json-pointer)
|
* [JSON Pointer](#json-pointer)
|
||||||
|
* [JSONPath](#jsonpath)
|
||||||
* [Error Handling](#error-handling)
|
* [Error Handling](#error-handling)
|
||||||
* [Error Handling Example](#error-handling-example)
|
* [Error Handling Example](#error-handling-example)
|
||||||
* [Exceptions](#exceptions)
|
* [Exceptions](#exceptions)
|
||||||
@@ -18,7 +19,7 @@ An overview of what you need to know to use simdjson, with examples.
|
|||||||
* [Padding and Temporary Copies](#padding-and-temporary-copies)
|
* [Padding and Temporary Copies](#padding-and-temporary-copies)
|
||||||
* [Performance Tips](#performance-tips)
|
* [Performance Tips](#performance-tips)
|
||||||
|
|
||||||
DOM vs On Demand
|
DOM vs On-Demand
|
||||||
----------------------------------------------
|
----------------------------------------------
|
||||||
|
|
||||||
The simdjson library offers two distinct approaches on how to access a JSON document. We support
|
The simdjson library offers two distinct approaches on how to access a JSON document. We support
|
||||||
@@ -257,6 +258,49 @@ for (dom::element car_element : cars) {
|
|||||||
}
|
}
|
||||||
```
|
```
|
||||||
|
|
||||||
|
JSONPath
|
||||||
|
------------
|
||||||
|
|
||||||
|
|
||||||
|
The simdjson library supports a subset of [JSONPath](https://datatracker.ietf.org/doc/html/draft-normington-jsonpath-00) through the `at_path()` method, allowing you to reach further into the document in a single call. The subset of JSONPath that is implemented is the subset that is trivially convertible into the JSON Pointer format, using `.` to access a field and `[]` to access a specific index.
|
||||||
|
|
||||||
|
Consider the following example:
|
||||||
|
|
||||||
|
```c++
|
||||||
|
auto cars_json = R"( [
|
||||||
|
{ "make": "Toyota", "model": "Camry", "year": 2018, "tire_pressure": [ 40.1, 39.9, 37.7, 40.4 ] },
|
||||||
|
{ "make": "Kia", "model": "Soul", "year": 2012, "tire_pressure": [ 30.1, 31.0, 28.6, 28.7 ] },
|
||||||
|
{ "make": "Toyota", "model": "Tercel", "year": 1999, "tire_pressure": [ 29.8, 30.0, 30.2, 30.5 ] }
|
||||||
|
] )"_padded;
|
||||||
|
dom::parser parser;
|
||||||
|
dom::element doc;
|
||||||
|
auto error = parser.parse(cars_json).get(doc);
|
||||||
|
if(error) { /*won't happen*/ }
|
||||||
|
double p;
|
||||||
|
error = doc.at_path("[0].tire_pressure[1]").get(p);
|
||||||
|
if(error) { /*won't happen*/ }
|
||||||
|
cout << p << endl; // Prints 39.9
|
||||||
|
```
|
||||||
|
|
||||||
|
|
||||||
|
We also support the `$` prefix. When you start a JSONPath expression with $, you are indicating that the path starts from the root of the JSON document. E.g.,
|
||||||
|
|
||||||
|
```c++
|
||||||
|
auto json = R"( { "c" :{ "foo": { "a": [ 10, 20, 30 ] }}, "d": { "foo2": { "a": [ 10, 20, 30 ] }} , "e": 120 })"_padded;
|
||||||
|
dom::parser parser;
|
||||||
|
dom::element doc;
|
||||||
|
auto error = parser.parse(json).get(doc);
|
||||||
|
if(error) { /*won't happen*/ }
|
||||||
|
dom::object obj;
|
||||||
|
error = doc.get_object().get(obj);
|
||||||
|
if(error) { /*won't happen*/ }
|
||||||
|
int64_t x;
|
||||||
|
error = obj.at_path("$[3].foo.a[1]").get(x);
|
||||||
|
if(error) { /*won't happen*/ }
|
||||||
|
if(x != 20) { /*won't happen*/ }
|
||||||
|
x = obj.at_path("$.d.foo2.a.2");
|
||||||
|
if(error) { /*won't happen*/ }
|
||||||
|
```
|
||||||
|
|
||||||
|
|
||||||
Error Handling
|
Error Handling
|
||||||
|
|||||||
+6
-1
@@ -102,7 +102,12 @@ remove almost entirely its cost and replaces it by the overhead of a thread, whi
|
|||||||
cheaper. Ain't that awesome!
|
cheaper. Ain't that awesome!
|
||||||
|
|
||||||
Thread support is only active if thread supported is detected in which case the macro
|
Thread support is only active if thread supported is detected in which case the macro
|
||||||
SIMDJSON_THREADS_ENABLED is set. Otherwise the library runs in single-thread mode.
|
SIMDJSON_THREADS_ENABLED is set. You can also manually pass `SIMDJSON_THREADS_ENABLED=1` flag
|
||||||
|
to the library. Otherwise the library runs in single-thread mode.
|
||||||
|
|
||||||
|
You should be consistent. If you link against the simdjson library built for multithreading
|
||||||
|
(i.e., with `SIMDJSON_THREADS_ENABLED`), then you should build your application with multithreading
|
||||||
|
system (setting `SIMDJSON_THREADS_ENABLED=1` and linking against a thread library).
|
||||||
|
|
||||||
A `document_stream` instance uses at most two threads: there is a main thread and a worker thread.
|
A `document_stream` instance uses at most two threads: there is a main thread and a worker thread.
|
||||||
|
|
||||||
|
|||||||
+30
-30
@@ -9,8 +9,8 @@ Whether we parse JSON or XML, or any other serialized format, there are relative
|
|||||||
- Another popular approach is the schema-based deserialization model.
|
- Another popular approach is the schema-based deserialization model.
|
||||||
|
|
||||||
We propose an approach that is as easy to use and often as flexible as the DOM approach, yet as fast and
|
We propose an approach that is as easy to use and often as flexible as the DOM approach, yet as fast and
|
||||||
efficient as the schema-based or event-based approaches. We call this new approach "On Demand". The
|
efficient as the schema-based or event-based approaches. We call this new approach "On-Demand". The
|
||||||
simdjson On Demand API offers a familiar, friendly DOM API and
|
simdjson On-Demand API offers a familiar, friendly DOM API and
|
||||||
provides the performance of just-in-time parsing on top of the simdjson superior performance.
|
provides the performance of just-in-time parsing on top of the simdjson superior performance.
|
||||||
|
|
||||||
To achieve ease of use, we mimicked the *form* of a traditional DOM API: you can iterate over
|
To achieve ease of use, we mimicked the *form* of a traditional DOM API: you can iterate over
|
||||||
@@ -18,7 +18,7 @@ arrays, look up fields in objects, and extract native values like `double`, `uin
|
|||||||
|
|
||||||
To achieve performance, we introduced some key limitations that make the DOM API *streaming*:
|
To achieve performance, we introduced some key limitations that make the DOM API *streaming*:
|
||||||
array/object iteration cannot be restarted, and string/number values can only be parsed once. If
|
array/object iteration cannot be restarted, and string/number values can only be parsed once. If
|
||||||
these limitations are acceptable to you, the On Demand API could help you write maintainable
|
these limitations are acceptable to you, the On-Demand API could help you write maintainable
|
||||||
applications with a computation efficiency that is difficult to surpass.
|
applications with a computation efficiency that is difficult to surpass.
|
||||||
|
|
||||||
A code example illustrates our API from a programmer's point of view:
|
A code example illustrates our API from a programmer's point of view:
|
||||||
@@ -72,24 +72,24 @@ This streaming approach means that unused fields and values are not parsed or
|
|||||||
converted, thus saving space and time. In our example, the `"name"`, `"followers_count"`,
|
converted, thus saving space and time. In our example, the `"name"`, `"followers_count"`,
|
||||||
and `"friends_count"` keys and matching values are skipped.
|
and `"friends_count"` keys and matching values are skipped.
|
||||||
|
|
||||||
Further, the On Demand API does not parse a value *at all* until you try to convert it (e.g., to `double`,
|
Further, the On-Demand API does not parse a value *at all* until you try to convert it (e.g., to `double`,
|
||||||
`int`, `string`, or `bool`). In our example, when accessing the key-value pair `"retweet_count": 82`, the parser
|
`int`, `string`, or `bool`). In our example, when accessing the key-value pair `"retweet_count": 82`, the parser
|
||||||
may not convert the pair of characters `82` to the binary integer 82. Because the programmer specifies the data
|
may not convert the pair of characters `82` to the binary integer 82. Because the programmer specifies the data
|
||||||
type, we avoid branch mispredictions related to data type determination and improve the performance.
|
type, we avoid branch mispredictions related to data type determination and improve the performance.
|
||||||
|
|
||||||
|
|
||||||
We expect users of an On Demand API to work in terms of a JSON dialect, which is a set of expectations and
|
We expect users of an On-Demand API to work in terms of a JSON dialect, which is a set of expectations and
|
||||||
specifications that come in addition to the [JSON specification](https://www.rfc-editor.org/rfc/rfc8259.txt).
|
specifications that come in addition to the [JSON specification](https://www.rfc-editor.org/rfc/rfc8259.txt).
|
||||||
The On Demand approach is designed around several principles:
|
The On-Demand approach is designed around several principles:
|
||||||
|
|
||||||
* **Streaming (\*):** It avoids preparsing values, keeping the memory usage and the latency down.
|
* **Streaming (\*):** It avoids preparsing values, keeping the memory usage and the latency down.
|
||||||
* **Forward-Only:** To prevent reiteration of the same values and to keep the number of variables down (literally), only a single index is maintained and everything uses it (even if you have nested for loops). This means when you are going through an array of arrays, for example, that the inner array loop will advance the index to the next comma, and the array can just pick it up and look at it.
|
* **Forward-Only:** To prevent reiteration of the same values and to keep the number of variables down (literally), only a single index is maintained and everything uses it (even if you have nested for loops). This means when you are going through an array of arrays, for example, that the inner array loop will advance the index to the next comma, and the array can just pick it up and look at it.
|
||||||
* **Natural Iteration:** A JSON array or object can be iterated with a normal C++ for loop. Nested arrays and objects are supported by nested for loops.
|
* **Natural Iteration:** A JSON array or object can be iterated with a normal C++ for loop. Nested arrays and objects are supported by nested for loops.
|
||||||
* **Use-Specific Parsing:** Parsing is always specific to the type required by the programmer. For example, if the programmer asks for an unsigned integer, we just start parsing digits. If there were no digits, we toss an error. There are even different parsers for `double`, `uint64_t` and `int64_t` values. This use-specific parsing avoids the branchiness of a generic "type switch," and makes the code more inlineable and compact.
|
* **Use-Specific Parsing:** Parsing is always specific to the type required by the programmer. For example, if the programmer asks for an unsigned integer, we just start parsing digits. If there were no digits, we toss an error. There are even different parsers for `double`, `uint64_t` and `int64_t` values. This use-specific parsing avoids the branchiness of a generic "type switch," and makes the code more inlineable and compact.
|
||||||
* **Validate What You Use:** On Demand deliberately validates the values you use and the structure leading to it, but nothing else. The goal is a guarantee that the value you asked for is the correct one and is not malformed: there must be no confusion over whether you got the right value.
|
* **Validate What You Use:** On-Demand deliberately validates the values you use and the structure leading to it, but nothing else. The goal is a guarantee that the value you asked for is the correct one and is not malformed: there must be no confusion over whether you got the right value.
|
||||||
|
|
||||||
|
|
||||||
To understand why On Demand is different, it is helpful to review the major
|
To understand why On-Demand is different, it is helpful to review the major
|
||||||
approaches to parsing and parser APIs in use today.
|
approaches to parsing and parser APIs in use today.
|
||||||
|
|
||||||
### DOM Parsers
|
### DOM Parsers
|
||||||
@@ -106,7 +106,7 @@ DOM tree is often easy enough that many users use the DOM as-is instead of creat
|
|||||||
their own custom data structures.
|
their own custom data structures.
|
||||||
|
|
||||||
The DOM approach was the only way to parse JSON documents up to version 0.6 of the simdjson library.
|
The DOM approach was the only way to parse JSON documents up to version 0.6 of the simdjson library.
|
||||||
Our DOM API looks similar to our On Demand example, except
|
Our DOM API looks similar to our On-Demand example, except
|
||||||
it calls `parse` instead of `iterate`:
|
it calls `parse` instead of `iterate`:
|
||||||
|
|
||||||
```c++
|
```c++
|
||||||
@@ -152,7 +152,7 @@ a tweet right now, or is this from some other place in the document
|
|||||||
entirely? Though an event-based approach may allow superior performance, it is demanding of the programmer
|
entirely? Though an event-based approach may allow superior performance, it is demanding of the programmer
|
||||||
who must efficiently keep track of its current state within the JSON input.
|
who must efficiently keep track of its current state within the JSON input.
|
||||||
|
|
||||||
The following is event-based example of the Twitter problem we have reviewed in the DOM and On Demand
|
The following is event-based example of the Twitter problem we have reviewed in the DOM and On-Demand
|
||||||
examples. To make it short enough to use as an example at all, it has heavily redacted: it only solves
|
examples. To make it short enough to use as an example at all, it has heavily redacted: it only solves
|
||||||
a part of the problem (does not get user.screen_name), it has bugs (it does not handle sub-objects
|
a part of the problem (does not get user.screen_name), it has bugs (it does not handle sub-objects
|
||||||
in a tweet at all), and it uses a theoretical, simple event-based API that minimizes ceremony.
|
in a tweet at all), and it uses a theoretical, simple event-based API that minimizes ceremony.
|
||||||
@@ -257,7 +257,7 @@ stress the branch prediction. Though branch predictors improve with each new gen
|
|||||||
the cost of branch mispredictions also tends to increase as pipelines expand, and the processors become
|
the cost of branch mispredictions also tends to increase as pipelines expand, and the processors become
|
||||||
able to schedule longer streams of instructions.
|
able to schedule longer streams of instructions.
|
||||||
|
|
||||||
On Demand parsing is tailor-made to solve this problem at the source, parsing values only after the
|
On-Demand parsing is tailor-made to solve this problem at the source, parsing values only after the
|
||||||
user declares their type by asking for a `double`, an `int`, a `string`, etc. It attempts to do so while
|
user declares their type by asking for a `double`, an `int`, a `string`, etc. It attempts to do so while
|
||||||
preserving most of the flexibility of DOM parsing.
|
preserving most of the flexibility of DOM parsing.
|
||||||
|
|
||||||
@@ -297,7 +297,7 @@ To help visualize the algorithm, we'll walk through the example C++ given at the
|
|||||||
|
|
||||||
Since this is the first time this parser has been used, `iterate()` first allocates internal
|
Since this is the first time this parser has been used, `iterate()` first allocates internal
|
||||||
parser buffers if this is the first time through. When reusing an existing parser, allocation
|
parser buffers if this is the first time through. When reusing an existing parser, allocation
|
||||||
only happens if the new document is bigger than internal buffers can handle. The On Demand
|
only happens if the new document is bigger than internal buffers can handle. The On-Demand
|
||||||
API only ever allocates memory in the `iterate()` function call.
|
API only ever allocates memory in the `iterate()` function call.
|
||||||
|
|
||||||
The simdjson library then preprocesses the JSON text at high speed, finding all tokens (i.e. the starting
|
The simdjson library then preprocesses the JSON text at high speed, finding all tokens (i.e. the starting
|
||||||
@@ -492,7 +492,7 @@ To help visualize the algorithm, we'll walk through the example C++ given at the
|
|||||||
Because of the cast to uint64_t, simdjson knows it's parsing an unsigned integer. This lets
|
Because of the cast to uint64_t, simdjson knows it's parsing an unsigned integer. This lets
|
||||||
us use a fast parser which *only* knows how to parse digits. It validates that it is an integer
|
us use a fast parser which *only* knows how to parse digits. It validates that it is an integer
|
||||||
by rejecting negative numbers, strings, and other values based on the fact that they are not the
|
by rejecting negative numbers, strings, and other values based on the fact that they are not the
|
||||||
digits 0-9. This type specificity is part of why parsing with on demand is so fast: you lose all
|
digits 0-9. This type specificity is part of why parsing with On-Demand is so fast: you lose all
|
||||||
the code that has to understand those other types.
|
the code that has to understand those other types.
|
||||||
|
|
||||||
The iterator is advanced to the `}`, and depth decreased back to 3 (root > statuses > tweet).
|
The iterator is advanced to the `}`, and depth decreased back to 3 (root > statuses > tweet).
|
||||||
@@ -597,7 +597,7 @@ To help visualize the algorithm, we'll walk through the example C++ given at the
|
|||||||
|
|
||||||
This means you can very efficiently do things like read a single value from a JSON file, or take
|
This means you can very efficiently do things like read a single value from a JSON file, or take
|
||||||
the top N, for example. It also means the things you don't use won't be fully validated. This is
|
the top N, for example. It also means the things you don't use won't be fully validated. This is
|
||||||
a general principle of On Demand: don't validate what you don't use. We still fully validate
|
a general principle of On-Demand: don't validate what you don't use. We still fully validate
|
||||||
values you do use, however, as well as the objects and arrays that lead to them, so that you can
|
values you do use, however, as well as the objects and arrays that lead to them, so that you can
|
||||||
be sure you get the information you need.
|
be sure you get the information you need.
|
||||||
|
|
||||||
@@ -654,7 +654,7 @@ for(auto field : doc.get_object()) {
|
|||||||
|
|
||||||
### Iteration Safety
|
### Iteration Safety
|
||||||
|
|
||||||
The On Demand API is powerful. To compensate, we add some safeguards to ensure that it can be used without fear
|
The On-Demand API is powerful. To compensate, we add some safeguards to ensure that it can be used without fear
|
||||||
in production systems:
|
in production systems:
|
||||||
|
|
||||||
- If the value fails to be parsed as one type, the program can try to parse it as something else until the program succeeds. Thus
|
- If the value fails to be parsed as one type, the program can try to parse it as something else until the program succeeds. Thus
|
||||||
@@ -667,7 +667,7 @@ in production systems:
|
|||||||
if it was `nullptr` but did not care what the actual value was--it will iterate. The destructor automates
|
if it was `nullptr` but did not care what the actual value was--it will iterate. The destructor automates
|
||||||
the iteration.
|
the iteration.
|
||||||
|
|
||||||
Some care is needed when using the On Demand API in scenarios where you need to access several sibling arrays or objects because
|
Some care is needed when using the On-Demand API in scenarios where you need to access several sibling arrays or objects because
|
||||||
only one object or array can be active at any one time. Let us consider the following example:
|
only one object or array can be active at any one time. Let us consider the following example:
|
||||||
|
|
||||||
```C++
|
```C++
|
||||||
@@ -709,36 +709,36 @@ A correct usage is given by the following example:
|
|||||||
}
|
}
|
||||||
```
|
```
|
||||||
|
|
||||||
### Benefits of the On Demand Approach
|
### Benefits of the On-Demand Approach
|
||||||
|
|
||||||
We expect that the On Demand approach has many of the performance benefits of the schema-based approach, while providing a flexibility that is similar to that of the DOM-based approach.
|
We expect that the On-Demand approach has many of the performance benefits of the schema-based approach, while providing a flexibility that is similar to that of the DOM-based approach.
|
||||||
|
|
||||||
* Faster than DOM in some cases. Reduced memory usage.
|
* Faster than DOM in some cases. Reduced memory usage.
|
||||||
* Straightforward, programmer-friendly interface (arrays and objects).
|
* Straightforward, programmer-friendly interface (arrays and objects).
|
||||||
* Highly expressive, beyond deserialization and pointer queries: many tasks can be accomplished with little code.
|
* Highly expressive, beyond deserialization and pointer queries: many tasks can be accomplished with little code.
|
||||||
|
|
||||||
### Limitations of the On Demand Approach
|
### Limitations of the On-Demand Approach
|
||||||
|
|
||||||
The On Demand approach has some limitations:
|
The On-Demand approach has some limitations:
|
||||||
|
|
||||||
* Because it operates in streaming mode, you only have access to the current element in the JSON document. Furthermore, the document is traversed in order so the code is sensitive to the order of the JSON nodes in the same manner as an event-based approach (e.g., SAX). (The one exception to this is field lookup, which is more *performant* when the order of lookups matches the order of fields in the document, but which will still work with out-of-order fields, with a performance hit.)
|
* Because it operates in streaming mode, you only have access to the current element in the JSON document. Furthermore, the document is traversed in order so the code is sensitive to the order of the JSON nodes in the same manner as an event-based approach (e.g., SAX). (The one exception to this is field lookup, which is more *performant* when the order of lookups matches the order of fields in the document, but which will still work with out-of-order fields, with a performance hit.)
|
||||||
* The On Demand approach is less safe than DOM: we only validate the components of the JSON document that are used and it is possible to begin ingesting an invalid document only to find out later that the document is invalid. Are you fine ingesting a large JSON document that starts with well formed JSON but ends with invalid JSON content?
|
* The On-Demand approach is less safe than DOM: we only validate the components of the JSON document that are used and it is possible to begin ingesting an invalid document only to find out later that the document is invalid. Are you fine ingesting a large JSON document that starts with well formed JSON but ends with invalid JSON content?
|
||||||
|
|
||||||
There are currently additional technical limitations which we expect to resolve in future releases of the simdjson library:
|
There are currently additional technical limitations which we expect to resolve in future releases of the simdjson library:
|
||||||
|
|
||||||
* The simdjson library offers runtime dispatching which allows you to compile one binary and have it run at full speed on different processors, taking advantage of the specific features of the processor. The On Demand API has limited runtime dispatch support. Under x64 systems, to fully benefit from the On Demand API, we recommend that you compile your code for a specific processor. E.g., if your processor supports AVX2 instructions, you should compile your binary executable with AVX2 instruction support (by using your compiler's commands). If you are sufficiently technically proficient, you can implement runtime dispatching within your application, by compiling your On Demand code for different processors.
|
* The simdjson library offers runtime dispatching which allows you to compile one binary and have it run at full speed on different processors, taking advantage of the specific features of the processor. The On-Demand API has limited runtime dispatch support. Under x64 systems, to fully benefit from the On-Demand API, we recommend that you compile your code for a specific processor. E.g., if your processor supports AVX2 instructions, you should compile your binary executable with AVX2 instruction support (by using your compiler's commands). If you are sufficiently technically proficient, you can implement runtime dispatching within your application, by compiling your On-Demand code for different processors.
|
||||||
* There is an initial phase which scans the entire document quickly, irrespective of the size of the document. We plan to break this phase into distinct steps for large files in a future release as we have done with other components of our API (e.g., `parse_many`).
|
* There is an initial phase which scans the entire document quickly, irrespective of the size of the document. We plan to break this phase into distinct steps for large files in a future release as we have done with other components of our API (e.g., `parse_many`).
|
||||||
|
|
||||||
### Applicability of the On Demand Approach
|
### Applicability of the On-Demand Approach
|
||||||
|
|
||||||
At this time we recommend the On Demand API in the following cases:
|
At this time we recommend the On-Demand API in the following cases:
|
||||||
|
|
||||||
1. The 64-bit hardware (CPU) used to run the software is known at compile time. If you need runtime dispatching because you cannot be certain of the hardware used to run your software, you will be better served with the core simdjson API. (This only applies to x64 (AMD/Intel). On 64-bit ARM hardware, runtime dispatching is unnecessary.)
|
1. The 64-bit hardware (CPU) used to run the software is known at compile time. If you need runtime dispatching because you cannot be certain of the hardware used to run your software, you will be better served with the core simdjson API. (This only applies to x64 (AMD/Intel). On 64-bit ARM hardware, runtime dispatching is unnecessary.)
|
||||||
2. The used parts of JSON files do not need to be validated and the layout of the nodes follows a strict JSON dialect. If you are receiving JSON from other systems, you might be better served with core simdjson API as it fully validates the JSON inputs and allows you to navigate through the document at will.
|
2. The used parts of JSON files do not need to be validated and the layout of the nodes follows a strict JSON dialect. If you are receiving JSON from other systems, you might be better served with core simdjson API as it fully validates the JSON inputs and allows you to navigate through the document at will.
|
||||||
3. Speed and efficiency are of the utmost importance. Keep in mind that the core simdjson API is highly efficient so adopting the On Demand API is not necessary for high efficiency.
|
3. Speed and efficiency are of the utmost importance. Keep in mind that the core simdjson API is highly efficient so adopting the On-Demand API is not necessary for high efficiency.
|
||||||
4. As a developer, you value a clean, flexible and maintainable API.
|
4. As a developer, you value a clean, flexible and maintainable API.
|
||||||
|
|
||||||
Good applications for the On Demand API might be:
|
Good applications for the On-Demand API might be:
|
||||||
|
|
||||||
* You are working from pre-existing large JSON files that have been vetted. You expect them to be well formed according to a known JSON dialect and to have a consistent layout. For example, you might be doing biomedical research or machine learning on top of static data dumps in JSON.
|
* You are working from pre-existing large JSON files that have been vetted. You expect them to be well formed according to a known JSON dialect and to have a consistent layout. For example, you might be doing biomedical research or machine learning on top of static data dumps in JSON.
|
||||||
* Both the generation and the consumption of JSON data is within your system. Your team controls both the software that produces the JSON and the software the parses it, your team knows and control the hardware. Thus you can fully test your system.
|
* Both the generation and the consumption of JSON data is within your system. Your team controls both the software that produces the JSON and the software the parses it, your team knows and control the hardware. Thus you can fully test your system.
|
||||||
@@ -746,13 +746,13 @@ Good applications for the On Demand API might be:
|
|||||||
|
|
||||||
## Checking Your CPU Selection (x64 systems)
|
## Checking Your CPU Selection (x64 systems)
|
||||||
|
|
||||||
The On Demand API uses advanced architecture-specific code for many common processors to make JSON preprocessing and string parsing faster. By default, however, most c++ compilers will compile to the least common denominator (since the program could theoretically be run anywhere). Since On Demand is inlined into your own code, it cannot always use these advanced versions unless the compiler is told to target them.
|
The On-Demand API uses advanced architecture-specific code for many common processors to make JSON preprocessing and string parsing faster. By default, however, most c++ compilers will compile to the least common denominator (since the program could theoretically be run anywhere). Since On-Demand is inlined into your own code, it cannot always use these advanced versions unless the compiler is told to target them.
|
||||||
|
|
||||||
On relevant systems, the On Demand API provides some support for runtime dispatching: that is, it will attempt to detect, at runtime, the instructions that your processor supports and optimize the code accordingly. However, it cannot always make full use of the features of your processor.
|
On relevant systems, the On-Demand API provides some support for runtime dispatching: that is, it will attempt to detect, at runtime, the instructions that your processor supports and optimize the code accordingly. However, it cannot always make full use of the features of your processor.
|
||||||
|
|
||||||
Some users wish to run at the best possible speed. Under recent Intel and AMD processors, these users should take additional steps to verify that their code is well optimized.
|
Some users wish to run at the best possible speed. Under recent Intel and AMD processors, these users should take additional steps to verify that their code is well optimized.
|
||||||
|
|
||||||
Given that the On Demand API offer limited runtime dispatching, it matters that your code is compiled against a specific CPU target. You should verify that the code is compiled against the target you expect. Thankfully, the simdjson library will tell you exactly what it detects as an implementation: `icelake` (AVX512 x64 processors), `haswell` (AVX2 x64 processors), `westmere` (SSE4 x64 processors), `arm64` (64-bit ARM), `ppc64` (64-bit POWER), `lasx` (LoongArch), `lsx` (LoongArch), `fallback` (others). Under x64 processors, many programmers will want to target `haswell` whereas under ARM, most programmers will want to target `arm64` (and it should do so automatically). The `fallback` is probably only good for testing purposes, not for deployment.
|
Given that the On-Demand API offer limited runtime dispatching, it matters that your code is compiled against a specific CPU target. You should verify that the code is compiled against the target you expect. Thankfully, the simdjson library will tell you exactly what it detects as an implementation: `icelake` (AVX512 x64 processors), `haswell` (AVX2 x64 processors), `westmere` (SSE4 x64 processors), `arm64` (64-bit ARM), `ppc64` (64-bit POWER), `lasx` (LoongArch), `lsx` (LoongArch), `fallback` (others). Under x64 processors, many programmers will want to target `haswell` whereas under ARM, most programmers will want to target `arm64` (and it should do so automatically). The `fallback` is probably only good for testing purposes, not for deployment.
|
||||||
|
|
||||||
```C++
|
```C++
|
||||||
std::cout << simdjson::builtin_implementation()->name() << std::endl;
|
std::cout << simdjson::builtin_implementation()->name() << std::endl;
|
||||||
@@ -777,6 +777,6 @@ In these examples, the `-march=haswell` flags targets a haswell processor and th
|
|||||||
|
|
||||||
Instead of specifying a specific microarchitecture, you can let your compiler do the work. The `-march=native` flags says "target the current computer," which is a reasonable default for many applications which both compile and run on the same processor.
|
Instead of specifying a specific microarchitecture, you can let your compiler do the work. The `-march=native` flags says "target the current computer," which is a reasonable default for many applications which both compile and run on the same processor.
|
||||||
|
|
||||||
Passing `-march=native` to the compiler may make On Demand faster by allowing it to use optimizations specific to your machine. You cannot do this, however, if you are compiling code that might be run on less advanced machines. That is, be mindful that when compiling with the `-march=native` flag, the resulting binary will run on the current system but may not run on other systems (e.g., on an old processor).
|
Passing `-march=native` to the compiler may make On-Demand faster by allowing it to use optimizations specific to your machine. You cannot do this, however, if you are compiling code that might be run on less advanced machines. That is, be mindful that when compiling with the `-march=native` flag, the resulting binary will run on the current system but may not run on other systems (e.g., on an old processor).
|
||||||
|
|
||||||
If you are compiling on an ARM or POWER system, you do not need to be concerned with CPU selection during compilation. The `-march=native` flag is useful for best performance on x64 (e.g., Intel) systems but it is generally unsupported on some platforms such as ARM (aarch64) or POWER.
|
If you are compiling on an ARM or POWER system, you do not need to be concerned with CPU selection during compilation. The `-march=native` flag is useful for best performance on x64 (e.g., Intel) systems but it is generally unsupported on some platforms such as ARM (aarch64) or POWER.
|
||||||
|
|||||||
+7
-2
@@ -103,7 +103,12 @@ cases, remove almost entirely its cost and replaces it by the overhead of a thre
|
|||||||
of magnitude cheaper. Ain't that awesome!
|
of magnitude cheaper. Ain't that awesome!
|
||||||
|
|
||||||
Thread support is only active if thread supported is detected in which case the macro
|
Thread support is only active if thread supported is detected in which case the macro
|
||||||
SIMDJSON_THREADS_ENABLED is set. Otherwise the library runs in single-thread mode.
|
SIMDJSON_THREADS_ENABLED is set. You can also manually pass `SIMDJSON_THREADS_ENABLED=1` flag
|
||||||
|
to the library. Otherwise the library runs in single-thread mode.
|
||||||
|
|
||||||
|
You should be consistent. If you link against the simdjson library built for multithreading
|
||||||
|
(i.e., with `SIMDJSON_THREADS_ENABLED`), then you should build your application with multithreading
|
||||||
|
system (setting `SIMDJSON_THREADS_ENABLED=1` and linking against a thread library).
|
||||||
|
|
||||||
A `document_stream` instance uses at most two threads: there is a main thread and a worker thread.
|
A `document_stream` instance uses at most two threads: there is a main thread and a worker thread.
|
||||||
You should expect the main thread to be fully occupied while the worker thread is partially busy
|
You should expect the main thread to be fully occupied while the worker thread is partially busy
|
||||||
@@ -125,7 +130,7 @@ Whitespace Characters:
|
|||||||
- **Nothing**
|
- **Nothing**
|
||||||
|
|
||||||
Some official formats **(non-exhaustive list)**:
|
Some official formats **(non-exhaustive list)**:
|
||||||
- [Newline-Delimited JSON (NDJSON)](http://ndjson.org/)
|
- [Newline-Delimited JSON (NDJSON)](https://github.com/ndjson/ndjson-spec)
|
||||||
- [JSON lines (JSONL)](http://jsonlines.org/)
|
- [JSON lines (JSONL)](http://jsonlines.org/)
|
||||||
- [Record separator-delimited JSON (RFC 7464)](https://tools.ietf.org/html/rfc7464) <- Not supported by JsonStream!
|
- [Record separator-delimited JSON (RFC 7464)](https://tools.ietf.org/html/rfc7464) <- Not supported by JsonStream!
|
||||||
- [More on Wikipedia...](https://en.wikipedia.org/wiki/JSON_streaming)
|
- [More on Wikipedia...](https://en.wikipedia.org/wiki/JSON_streaming)
|
||||||
|
|||||||
+109
-1
@@ -14,6 +14,7 @@ testing and get the best performance.
|
|||||||
* [Number parsing](#number-parsing)
|
* [Number parsing](#number-parsing)
|
||||||
* [Visual Studio](#visual-studio)
|
* [Visual Studio](#visual-studio)
|
||||||
* [Power Usage and Downclocking](#power-usage-and-downclocking)
|
* [Power Usage and Downclocking](#power-usage-and-downclocking)
|
||||||
|
* [Free Padding](#free-padding)
|
||||||
|
|
||||||
|
|
||||||
NDEBUG directive
|
NDEBUG directive
|
||||||
@@ -74,7 +75,7 @@ or simply
|
|||||||
Server Loops: Long-Running Processes and Memory Capacity
|
Server Loops: Long-Running Processes and Memory Capacity
|
||||||
---------------------------------
|
---------------------------------
|
||||||
|
|
||||||
The On Demand approach also automatically expands its memory capacity when larger documents are parsed. However, for longer processes where very large files are processed (such as server loops), this capacity is not resized down. On Demand also lets you adjust the maximal capacity that the parser can process:
|
The On-Demand approach also automatically expands its memory capacity when larger documents are parsed. However, for longer processes where very large files are processed (such as server loops), this capacity is not resized down. On-Demand also lets you adjust the maximal capacity that the parser can process:
|
||||||
|
|
||||||
* You can set an upper bound (*max_capacity*) when construction the parser:
|
* You can set an upper bound (*max_capacity*) when construction the parser:
|
||||||
```C++
|
```C++
|
||||||
@@ -179,3 +180,110 @@ The simdjson library does not generally make use of heavy 256-bit instructions.
|
|||||||
the macro `SIMDJSON_AVX512_ALLOWED` to `0` in C++ prior to importing the headers.
|
the macro `SIMDJSON_AVX512_ALLOWED` to `0` in C++ prior to importing the headers.
|
||||||
|
|
||||||
You may still be worried about which SIMD instruction set is used by simdjson. Thankfully, [you can always determine and change which architecture-specific implementation is used](implementation-selection.md) by simdjson. Thus even if your CPU supports AVX2, you do not need to use AVX2. You are in control.
|
You may still be worried about which SIMD instruction set is used by simdjson. Thankfully, [you can always determine and change which architecture-specific implementation is used](implementation-selection.md) by simdjson. Thus even if your CPU supports AVX2, you do not need to use AVX2. You are in control.
|
||||||
|
|
||||||
|
|
||||||
|
Free Padding
|
||||||
|
-------
|
||||||
|
|
||||||
|
For performance reasons, the simdjson library requires that the JSON input contain at least
|
||||||
|
`simdjson::SIMDJSON_PADDING` bytes at the end of the stream. The value `simdjson::SIMDJSON_PADDING` is
|
||||||
|
small (e.g., 64 bytes). On modern systems, you can safely read beyond an allocated buffers,
|
||||||
|
as long as you remain within an allocated page. Pages on modern systems span at least 4 kilobytes,
|
||||||
|
but can be significantly larger. E.g., Apple systems favour pages spanning 16 kilobytes.
|
||||||
|
|
||||||
|
In effect, it means that you can almost always read a few bytes beyond your current buffer---without
|
||||||
|
allocating extra memory. However, tools such as valgrind or memory sanitizers will flag such behavior as unsafe.
|
||||||
|
Nevertheless, you can still make sure of this capability in your code if you are an expert
|
||||||
|
programmer and you are willing to silence sanitizer warnings. The following code provides
|
||||||
|
a portable example.
|
||||||
|
|
||||||
|
|
||||||
|
The conditional compilation checks for the `_MSC_VER` macro (indicating Microsoft Visual Studio)
|
||||||
|
and includes platform-specific headers accordingly.
|
||||||
|
The `page_size()` function determines the default size of a memory page in bytes on the system.
|
||||||
|
On Windows (when `_WIN32` is defined), it uses `GetSystemInfo()` to retrieve system information and obtain the page size.
|
||||||
|
On other platforms (non-Windows), it uses `sysconf(_SC_PAGESIZE)` to get the page size.
|
||||||
|
The function returns the page size.
|
||||||
|
The `need_allocation()` function checks whether the buffer (given by `buf`) plus the specified length (`len`) is near a page boundary.
|
||||||
|
If the buffer extends beyond the current page when padded by `simdjson::SIMDJSON_PADDING`, it returns true, indicating that reallocation is needed.
|
||||||
|
Otherwise, it returns false.
|
||||||
|
The `get_padded_string_view()` creates a `padded_string_view` from the input buffer.
|
||||||
|
If reallocation is needed (unlikely case), it allocates a new padded_string and assigns it to `jsonbuffer`.
|
||||||
|
Otherwise (very likely), it creates a `padded_string_view` directly from the buffer.
|
||||||
|
The `simdjson::SIMDJSON_PADDING` ensures that there is additional padding for parsing efficiency.
|
||||||
|
The calling code just needs to provide `jsonbuffer` (an instance of `simdjson::padded_string`)
|
||||||
|
and pass `get_padded_string_view(buf, len, jsonbuffer)` to `parser.iterate`. Most of the time,
|
||||||
|
this code will not allocate new memory.
|
||||||
|
|
||||||
|
|
||||||
|
```cpp
|
||||||
|
#ifdef _WIN32
|
||||||
|
#include <windows.h>
|
||||||
|
#include <sysinfoapi.h>
|
||||||
|
#else
|
||||||
|
#include <unistd.h>
|
||||||
|
#endif
|
||||||
|
#include "simdjson.h"
|
||||||
|
#include <cstdio>
|
||||||
|
|
||||||
|
// Returns the default size of the page in bytes on this system.
|
||||||
|
long page_size() {
|
||||||
|
#ifdef _WIN32
|
||||||
|
SYSTEM_INFO sysInfo;
|
||||||
|
GetSystemInfo(&sysInfo);
|
||||||
|
long pagesize = sysInfo.dwPageSize;
|
||||||
|
#else
|
||||||
|
long pagesize = sysconf(_SC_PAGESIZE);
|
||||||
|
#endif
|
||||||
|
return pagesize;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Returns true if the buffer + len + simdjson::SIMDJSON_PADDING crosses the
|
||||||
|
// page boundary.
|
||||||
|
bool need_allocation(const char *buf, size_t len) {
|
||||||
|
return ((reinterpret_cast<uintptr_t>(buf + len - 1) % page_size())
|
||||||
|
+ simdjson::SIMDJSON_PADDING > static_cast<uintptr_t>(page_size()));
|
||||||
|
}
|
||||||
|
|
||||||
|
simdjson::padded_string_view
|
||||||
|
get_padded_string_view(const char *buf, size_t len,
|
||||||
|
simdjson::padded_string &jsonbuffer) {
|
||||||
|
if (need_allocation(buf, len)) { // unlikely case
|
||||||
|
jsonbuffer = simdjson::padded_string(buf, len);
|
||||||
|
return jsonbuffer;
|
||||||
|
} else { // no reallcation needed (very likely)
|
||||||
|
return simdjson::padded_string_view(buf, len,
|
||||||
|
len + simdjson::SIMDJSON_PADDING);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
int main() {
|
||||||
|
printf("page_size: %ld\n", page_size());
|
||||||
|
const char *jsonpoiner = R"(
|
||||||
|
{
|
||||||
|
"key": "value"
|
||||||
|
}
|
||||||
|
)";
|
||||||
|
size_t len = strlen(jsonpoiner);
|
||||||
|
simdjson::padded_string jsonbuffer; // only allocate if needed
|
||||||
|
simdjson::ondemand::parser parser;
|
||||||
|
simdjson::ondemand::document doc;
|
||||||
|
simdjson::error_code error =
|
||||||
|
parser.iterate(get_padded_string_view(jsonpoiner, len, jsonbuffer))
|
||||||
|
.get(doc);
|
||||||
|
if (error) {
|
||||||
|
printf("error: %s\n", simdjson::error_message(error));
|
||||||
|
return EXIT_FAILURE;
|
||||||
|
}
|
||||||
|
std::string_view value;
|
||||||
|
error = doc["key"].get_string().get(value);
|
||||||
|
if (error) {
|
||||||
|
return EXIT_FAILURE;
|
||||||
|
}
|
||||||
|
printf("Value: \"%.*s\"\n", (int)value.size(), value.data());
|
||||||
|
if (value != "value") {
|
||||||
|
return EXIT_FAILURE;
|
||||||
|
}
|
||||||
|
return EXIT_SUCCESS;
|
||||||
|
}
|
||||||
|
```
|
||||||
@@ -25,7 +25,7 @@ IF(${CMAKE_SYSTEM_NAME} MATCHES "Linux")
|
|||||||
add_quickstart_test(quickstart2_noexceptions quickstart2_noexceptions.cpp NO_EXCEPTIONS LABELS acceptance)
|
add_quickstart_test(quickstart2_noexceptions quickstart2_noexceptions.cpp NO_EXCEPTIONS LABELS acceptance)
|
||||||
add_quickstart_test(quickstart2_noexceptions11 quickstart2_noexceptions.cpp NO_EXCEPTIONS CXX_STANDARD c++11)
|
add_quickstart_test(quickstart2_noexceptions11 quickstart2_noexceptions.cpp NO_EXCEPTIONS CXX_STANDARD c++11)
|
||||||
|
|
||||||
# On Demand Quick Start
|
# On-Demand Quick Start
|
||||||
if (SIMDJSON_EXCEPTIONS)
|
if (SIMDJSON_EXCEPTIONS)
|
||||||
add_quickstart_test(quickstart_ondemand quickstart_ondemand.cpp LABELS quickstart_ondemand acceptance)
|
add_quickstart_test(quickstart_ondemand quickstart_ondemand.cpp LABELS quickstart_ondemand acceptance)
|
||||||
add_quickstart_test(quickstart_ondemand11 quickstart_ondemand.cpp CXX_STANDARD c++11 LABELS quickstart_ondemand acceptance)
|
add_quickstart_test(quickstart_ondemand11 quickstart_ondemand.cpp CXX_STANDARD c++11 LABELS quickstart_ondemand acceptance)
|
||||||
|
|||||||
@@ -35,6 +35,12 @@ simdjson_inline uint64_t clear_lowest_bit(uint64_t input_num) {
|
|||||||
return input_num & (input_num-1);
|
return input_num & (input_num-1);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// We sometimes call leading_zeroes on inputs that are zero,
|
||||||
|
// but the algorithms do not end up using the returned value.
|
||||||
|
// Sadly, sanitizers are not smart enough to figure it out.
|
||||||
|
// Applies only when SIMDJSON_PREFER_REVERSE_BITS is defined and true.
|
||||||
|
// (See below.)
|
||||||
|
SIMDJSON_NO_SANITIZE_UNDEFINED
|
||||||
/* result might be undefined when input_num is zero */
|
/* result might be undefined when input_num is zero */
|
||||||
simdjson_inline int leading_zeroes(uint64_t input_num) {
|
simdjson_inline int leading_zeroes(uint64_t input_num) {
|
||||||
#ifdef SIMDJSON_REGULAR_VISUAL_STUDIO
|
#ifdef SIMDJSON_REGULAR_VISUAL_STUDIO
|
||||||
|
|||||||
@@ -9,10 +9,10 @@
|
|||||||
|
|
||||||
#include <cstring>
|
#include <cstring>
|
||||||
|
|
||||||
#if _M_ARM64
|
#if SIMDJSON_REGULAR_VISUAL_STUDIO && SIMDJSON_IS_ARM64
|
||||||
// __umulh requires intrin.h
|
// __umulh requires intrin.h
|
||||||
#include <intrin.h>
|
#include <intrin.h>
|
||||||
#endif // _M_ARM64
|
#endif // SIMDJSON_REGULAR_VISUAL_STUDIO && SIMDJSON_IS_ARM64
|
||||||
|
|
||||||
namespace simdjson {
|
namespace simdjson {
|
||||||
namespace arm64 {
|
namespace arm64 {
|
||||||
@@ -32,13 +32,13 @@ static simdjson_inline uint32_t parse_eight_digits_unrolled(const uint8_t *chars
|
|||||||
simdjson_inline internal::value128 full_multiplication(uint64_t value1, uint64_t value2) {
|
simdjson_inline internal::value128 full_multiplication(uint64_t value1, uint64_t value2) {
|
||||||
internal::value128 answer;
|
internal::value128 answer;
|
||||||
#if SIMDJSON_REGULAR_VISUAL_STUDIO || SIMDJSON_IS_32BITS
|
#if SIMDJSON_REGULAR_VISUAL_STUDIO || SIMDJSON_IS_32BITS
|
||||||
#ifdef _M_ARM64
|
#if SIMDJSON_IS_ARM64
|
||||||
// ARM64 has native support for 64-bit multiplications, no need to emultate
|
// ARM64 has native support for 64-bit multiplications, no need to emultate
|
||||||
answer.high = __umulh(value1, value2);
|
answer.high = __umulh(value1, value2);
|
||||||
answer.low = value1 * value2;
|
answer.low = value1 * value2;
|
||||||
#else
|
#else
|
||||||
answer.low = _umul128(value1, value2, &answer.high); // _umul128 not available on ARM64
|
answer.low = _umul128(value1, value2, &answer.high); // _umul128 not available on ARM64
|
||||||
#endif // _M_ARM64
|
#endif // SIMDJSON_IS_ARM64
|
||||||
#else // SIMDJSON_REGULAR_VISUAL_STUDIO || SIMDJSON_IS_32BITS
|
#else // SIMDJSON_REGULAR_VISUAL_STUDIO || SIMDJSON_IS_32BITS
|
||||||
__uint128_t r = (static_cast<__uint128_t>(value1)) * value2;
|
__uint128_t r = (static_cast<__uint128_t>(value1)) * value2;
|
||||||
answer.low = uint64_t(r);
|
answer.low = uint64_t(r);
|
||||||
@@ -51,6 +51,12 @@ simdjson_inline internal::value128 full_multiplication(uint64_t value1, uint64_t
|
|||||||
} // namespace arm64
|
} // namespace arm64
|
||||||
} // namespace simdjson
|
} // namespace simdjson
|
||||||
|
|
||||||
|
#ifndef SIMDJSON_SWAR_NUMBER_PARSING
|
||||||
|
#if SIMDJSON_IS_BIG_ENDIAN
|
||||||
|
#define SIMDJSON_SWAR_NUMBER_PARSING 0
|
||||||
|
#else
|
||||||
#define SIMDJSON_SWAR_NUMBER_PARSING 1
|
#define SIMDJSON_SWAR_NUMBER_PARSING 1
|
||||||
|
#endif
|
||||||
|
#endif
|
||||||
|
|
||||||
#endif // SIMDJSON_ARM64_NUMBERPARSING_DEFS_H
|
#endif // SIMDJSON_ARM64_NUMBERPARSING_DEFS_H
|
||||||
|
|||||||
@@ -134,7 +134,7 @@ namespace {
|
|||||||
tmp = vpaddq_u8(tmp, tmp);
|
tmp = vpaddq_u8(tmp, tmp);
|
||||||
return vgetq_lane_u16(vreinterpretq_u16_u8(tmp), 0);
|
return vgetq_lane_u16(vreinterpretq_u16_u8(tmp), 0);
|
||||||
}
|
}
|
||||||
simdjson_inline bool any() const { return vmaxvq_u8(*this) != 0; }
|
simdjson_inline bool any() const { return vmaxvq_u32(vreinterpretq_u32_u8(*this)) != 0; }
|
||||||
};
|
};
|
||||||
|
|
||||||
// Unsigned bytes
|
// Unsigned bytes
|
||||||
|
|||||||
@@ -50,6 +50,8 @@ double from_chars(const char *first, const char* end) noexcept;
|
|||||||
#define SIMDJSON_ISALIGNED_N(ptr, n) (((uintptr_t)(ptr) & ((n)-1)) == 0)
|
#define SIMDJSON_ISALIGNED_N(ptr, n) (((uintptr_t)(ptr) & ((n)-1)) == 0)
|
||||||
|
|
||||||
#if SIMDJSON_REGULAR_VISUAL_STUDIO
|
#if SIMDJSON_REGULAR_VISUAL_STUDIO
|
||||||
|
// We could use [[deprecated]] but it requires C++14
|
||||||
|
#define simdjson_deprecated __declspec(deprecated)
|
||||||
|
|
||||||
#define simdjson_really_inline __forceinline
|
#define simdjson_really_inline __forceinline
|
||||||
#define simdjson_never_inline __declspec(noinline)
|
#define simdjson_never_inline __declspec(noinline)
|
||||||
@@ -88,6 +90,8 @@ double from_chars(const char *first, const char* end) noexcept;
|
|||||||
#define SIMDJSON_POP_DISABLE_UNUSED_WARNINGS
|
#define SIMDJSON_POP_DISABLE_UNUSED_WARNINGS
|
||||||
|
|
||||||
#else // SIMDJSON_REGULAR_VISUAL_STUDIO
|
#else // SIMDJSON_REGULAR_VISUAL_STUDIO
|
||||||
|
// We could use [[deprecated]] but it requires C++14
|
||||||
|
#define simdjson_deprecated __attribute__((deprecated))
|
||||||
|
|
||||||
#define simdjson_really_inline inline __attribute__((always_inline))
|
#define simdjson_really_inline inline __attribute__((always_inline))
|
||||||
#define simdjson_never_inline inline __attribute__((noinline))
|
#define simdjson_never_inline inline __attribute__((noinline))
|
||||||
|
|||||||
@@ -13,6 +13,16 @@
|
|||||||
#endif
|
#endif
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
|
// C++ 23
|
||||||
|
#if !defined(SIMDJSON_CPLUSPLUS23) && (SIMDJSON_CPLUSPLUS >= 202302L)
|
||||||
|
#define SIMDJSON_CPLUSPLUS23 1
|
||||||
|
#endif
|
||||||
|
|
||||||
|
// C++ 20
|
||||||
|
#if !defined(SIMDJSON_CPLUSPLUS20) && (SIMDJSON_CPLUSPLUS >= 202002L)
|
||||||
|
#define SIMDJSON_CPLUSPLUS20 1
|
||||||
|
#endif
|
||||||
|
|
||||||
// C++ 17
|
// C++ 17
|
||||||
#if !defined(SIMDJSON_CPLUSPLUS17) && (SIMDJSON_CPLUSPLUS >= 201703L)
|
#if !defined(SIMDJSON_CPLUSPLUS17) && (SIMDJSON_CPLUSPLUS >= 201703L)
|
||||||
#define SIMDJSON_CPLUSPLUS17 1
|
#define SIMDJSON_CPLUSPLUS17 1
|
||||||
|
|||||||
@@ -7,6 +7,7 @@
|
|||||||
#include "simdjson/dom/array.h"
|
#include "simdjson/dom/array.h"
|
||||||
#include "simdjson/dom/element.h"
|
#include "simdjson/dom/element.h"
|
||||||
#include "simdjson/error-inl.h"
|
#include "simdjson/error-inl.h"
|
||||||
|
#include "simdjson/jsonpathutil.h"
|
||||||
#include "simdjson/internal/tape_ref-inl.h"
|
#include "simdjson/internal/tape_ref-inl.h"
|
||||||
|
|
||||||
#include <limits>
|
#include <limits>
|
||||||
@@ -44,6 +45,13 @@ inline simdjson_result<dom::element> simdjson_result<dom::array>::at_pointer(std
|
|||||||
if (error()) { return error(); }
|
if (error()) { return error(); }
|
||||||
return first.at_pointer(json_pointer);
|
return first.at_pointer(json_pointer);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
inline simdjson_result<dom::element> simdjson_result<dom::array>::at_path(std::string_view json_path) const noexcept {
|
||||||
|
auto json_pointer = json_path_to_pointer_conversion(json_path);
|
||||||
|
if (json_pointer == "-1") { return INVALID_JSON_POINTER; }
|
||||||
|
return at_pointer(json_pointer);
|
||||||
|
}
|
||||||
|
|
||||||
inline simdjson_result<dom::element> simdjson_result<dom::array>::at(size_t index) const noexcept {
|
inline simdjson_result<dom::element> simdjson_result<dom::array>::at(size_t index) const noexcept {
|
||||||
if (error()) { return error(); }
|
if (error()) { return error(); }
|
||||||
return first.at(index);
|
return first.at(index);
|
||||||
@@ -113,6 +121,12 @@ inline simdjson_result<element> array::at_pointer(std::string_view json_pointer)
|
|||||||
return child;
|
return child;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
inline simdjson_result<element> array::at_path(std::string_view json_path) const noexcept {
|
||||||
|
auto json_pointer = json_path_to_pointer_conversion(json_path);
|
||||||
|
if (json_pointer == "-1") { return INVALID_JSON_POINTER; }
|
||||||
|
return at_pointer(json_pointer);
|
||||||
|
}
|
||||||
|
|
||||||
inline simdjson_result<element> array::at(size_t index) const noexcept {
|
inline simdjson_result<element> array::at(size_t index) const noexcept {
|
||||||
SIMDJSON_DEVELOPMENT_ASSERT(tape.usable()); // https://github.com/simdjson/simdjson/issues/1914
|
SIMDJSON_DEVELOPMENT_ASSERT(tape.usable()); // https://github.com/simdjson/simdjson/issues/1914
|
||||||
size_t i=0;
|
size_t i=0;
|
||||||
@@ -123,6 +137,10 @@ inline simdjson_result<element> array::at(size_t index) const noexcept {
|
|||||||
return INDEX_OUT_OF_BOUNDS;
|
return INDEX_OUT_OF_BOUNDS;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
inline array::operator element() const noexcept {
|
||||||
|
return element(tape);
|
||||||
|
}
|
||||||
|
|
||||||
//
|
//
|
||||||
// array::iterator inline implementation
|
// array::iterator inline implementation
|
||||||
//
|
//
|
||||||
|
|||||||
@@ -108,6 +108,21 @@ public:
|
|||||||
*/
|
*/
|
||||||
inline simdjson_result<element> at_pointer(std::string_view json_pointer) const noexcept;
|
inline simdjson_result<element> at_pointer(std::string_view json_pointer) const noexcept;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Get the value associated with the given JSONPath expression. We only support
|
||||||
|
* JSONPath queries that trivially convertible to JSON Pointer queries: key
|
||||||
|
* names and array indices.
|
||||||
|
*
|
||||||
|
* https://datatracker.ietf.org/doc/html/draft-normington-jsonpath-00
|
||||||
|
*
|
||||||
|
* @return The value associated with the given JSONPath expression, or:
|
||||||
|
* - INVALID_JSON_POINTER if the JSONPath to JSON Pointer conversion fails
|
||||||
|
* - NO_SUCH_FIELD if a field does not exist in an object
|
||||||
|
* - INDEX_OUT_OF_BOUNDS if an array index is larger than an array length
|
||||||
|
* - INCORRECT_TYPE if a non-integer is used to access an array
|
||||||
|
*/
|
||||||
|
inline simdjson_result<element> at_path(std::string_view json_path) const noexcept;
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Get the value at the given index. This function has linear-time complexity and
|
* Get the value at the given index. This function has linear-time complexity and
|
||||||
* is equivalent to the following:
|
* is equivalent to the following:
|
||||||
@@ -126,6 +141,11 @@ public:
|
|||||||
*/
|
*/
|
||||||
inline simdjson_result<element> at(size_t index) const noexcept;
|
inline simdjson_result<element> at(size_t index) const noexcept;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Implicitly convert object to element
|
||||||
|
*/
|
||||||
|
inline operator element() const noexcept;
|
||||||
|
|
||||||
private:
|
private:
|
||||||
simdjson_inline array(const internal::tape_ref &tape) noexcept;
|
simdjson_inline array(const internal::tape_ref &tape) noexcept;
|
||||||
internal::tape_ref tape;
|
internal::tape_ref tape;
|
||||||
@@ -147,6 +167,7 @@ public:
|
|||||||
simdjson_inline simdjson_result(error_code error) noexcept; ///< @private
|
simdjson_inline simdjson_result(error_code error) noexcept; ///< @private
|
||||||
|
|
||||||
inline simdjson_result<dom::element> at_pointer(std::string_view json_pointer) const noexcept;
|
inline simdjson_result<dom::element> at_pointer(std::string_view json_pointer) const noexcept;
|
||||||
|
inline simdjson_result<dom::element> at_path(std::string_view json_path) const noexcept;
|
||||||
inline simdjson_result<dom::element> at(size_t index) const noexcept;
|
inline simdjson_result<dom::element> at(size_t index) const noexcept;
|
||||||
|
|
||||||
#if SIMDJSON_EXCEPTIONS
|
#if SIMDJSON_EXCEPTIONS
|
||||||
|
|||||||
@@ -223,7 +223,11 @@ simdjson_inline std::string_view document_stream::iterator::source() const noexc
|
|||||||
return std::string_view(start, next_doc_index - current_index() + 1);
|
return std::string_view(start, next_doc_index - current_index() + 1);
|
||||||
} else {
|
} else {
|
||||||
size_t next_doc_index = stream->batch_start + stream->parser->implementation->structural_indexes[stream->parser->implementation->next_structural_index];
|
size_t next_doc_index = stream->batch_start + stream->parser->implementation->structural_indexes[stream->parser->implementation->next_structural_index];
|
||||||
return std::string_view(reinterpret_cast<const char*>(stream->buf) + current_index(), next_doc_index - current_index() - 1);
|
size_t svlen = next_doc_index - current_index();
|
||||||
|
while(svlen > 1 && (std::isspace(start[svlen-1]) || start[svlen-1] == '\0')) {
|
||||||
|
svlen--;
|
||||||
|
}
|
||||||
|
return std::string_view(start, svlen);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -9,6 +9,7 @@
|
|||||||
|
|
||||||
#include "simdjson/dom/object-inl.h"
|
#include "simdjson/dom/object-inl.h"
|
||||||
#include "simdjson/error-inl.h"
|
#include "simdjson/error-inl.h"
|
||||||
|
#include "simdjson/jsonpathutil.h"
|
||||||
|
|
||||||
#include <ostream>
|
#include <ostream>
|
||||||
#include <limits>
|
#include <limits>
|
||||||
@@ -122,6 +123,11 @@ simdjson_inline simdjson_result<dom::element> simdjson_result<dom::element>::at_
|
|||||||
if (error()) { return error(); }
|
if (error()) { return error(); }
|
||||||
return first.at_pointer(json_pointer);
|
return first.at_pointer(json_pointer);
|
||||||
}
|
}
|
||||||
|
simdjson_inline simdjson_result<dom::element> simdjson_result<dom::element>::at_path(const std::string_view json_path) const noexcept {
|
||||||
|
auto json_pointer = json_path_to_pointer_conversion(json_path);
|
||||||
|
if (json_pointer == "-1") { return INVALID_JSON_POINTER; }
|
||||||
|
return at_pointer(json_pointer);
|
||||||
|
}
|
||||||
#ifndef SIMDJSON_DISABLE_DEPRECATED_API
|
#ifndef SIMDJSON_DISABLE_DEPRECATED_API
|
||||||
[[deprecated("For standard compliance, use at_pointer instead, and prefix your pointers with a slash '/', see RFC6901 ")]]
|
[[deprecated("For standard compliance, use at_pointer instead, and prefix your pointers with a slash '/', see RFC6901 ")]]
|
||||||
simdjson_inline simdjson_result<dom::element> simdjson_result<dom::element>::at(const std::string_view json_pointer) const noexcept {
|
simdjson_inline simdjson_result<dom::element> simdjson_result<dom::element>::at(const std::string_view json_pointer) const noexcept {
|
||||||
@@ -375,6 +381,23 @@ inline simdjson_result<element> element::operator[](const char *key) const noexc
|
|||||||
return at_key(key);
|
return at_key(key);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
inline bool is_pointer_well_formed(std::string_view json_pointer) noexcept {
|
||||||
|
if (simdjson_unlikely(json_pointer[0] != '/')) {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
size_t escape = json_pointer.find('~');
|
||||||
|
if (escape == std::string_view::npos) {
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
if (escape == json_pointer.size() - 1) {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
if (json_pointer[escape + 1] != '0' && json_pointer[escape + 1] != '1') {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
inline simdjson_result<element> element::at_pointer(std::string_view json_pointer) const noexcept {
|
inline simdjson_result<element> element::at_pointer(std::string_view json_pointer) const noexcept {
|
||||||
SIMDJSON_DEVELOPMENT_ASSERT(tape.usable()); // https://github.com/simdjson/simdjson/issues/1914
|
SIMDJSON_DEVELOPMENT_ASSERT(tape.usable()); // https://github.com/simdjson/simdjson/issues/1914
|
||||||
switch (tape.tape_ref_type()) {
|
switch (tape.tape_ref_type()) {
|
||||||
@@ -383,7 +406,10 @@ inline simdjson_result<element> element::at_pointer(std::string_view json_pointe
|
|||||||
case internal::tape_type::START_ARRAY:
|
case internal::tape_type::START_ARRAY:
|
||||||
return array(tape).at_pointer(json_pointer);
|
return array(tape).at_pointer(json_pointer);
|
||||||
default: {
|
default: {
|
||||||
if(!json_pointer.empty()) { // a non-empty string is invalid on an atom
|
if (!json_pointer.empty()) { // a non-empty string can be invalid, or accessing a primitive (issue 2154)
|
||||||
|
if (is_pointer_well_formed(json_pointer)) {
|
||||||
|
return NO_SUCH_FIELD;
|
||||||
|
}
|
||||||
return INVALID_JSON_POINTER;
|
return INVALID_JSON_POINTER;
|
||||||
}
|
}
|
||||||
// an empty string means that we return the current node
|
// an empty string means that we return the current node
|
||||||
@@ -392,6 +418,11 @@ inline simdjson_result<element> element::at_pointer(std::string_view json_pointe
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
inline simdjson_result<element> element::at_path(std::string_view json_path) const noexcept {
|
||||||
|
auto json_pointer = json_path_to_pointer_conversion(json_path);
|
||||||
|
if (json_pointer == "-1") { return INVALID_JSON_POINTER; }
|
||||||
|
return at_pointer(json_pointer);
|
||||||
|
}
|
||||||
#ifndef SIMDJSON_DISABLE_DEPRECATED_API
|
#ifndef SIMDJSON_DISABLE_DEPRECATED_API
|
||||||
[[deprecated("For standard compliance, use at_pointer instead, and prefix your pointers with a slash '/', see RFC6901 ")]]
|
[[deprecated("For standard compliance, use at_pointer instead, and prefix your pointers with a slash '/', see RFC6901 ")]]
|
||||||
inline simdjson_result<element> element::at(std::string_view json_pointer) const noexcept {
|
inline simdjson_result<element> element::at(std::string_view json_pointer) const noexcept {
|
||||||
|
|||||||
@@ -397,6 +397,21 @@ public:
|
|||||||
*/
|
*/
|
||||||
inline simdjson_result<element> at_pointer(const std::string_view json_pointer) const noexcept;
|
inline simdjson_result<element> at_pointer(const std::string_view json_pointer) const noexcept;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Get the value associated with the given JSONPath expression. We only support
|
||||||
|
* JSONPath queries that trivially convertible to JSON Pointer queries: key
|
||||||
|
* names and array indices.
|
||||||
|
*
|
||||||
|
* https://datatracker.ietf.org/doc/html/draft-normington-jsonpath-00
|
||||||
|
*
|
||||||
|
* @return The value associated with the given JSONPath expression, or:
|
||||||
|
* - INVALID_JSON_POINTER if the JSONPath to JSON Pointer conversion fails
|
||||||
|
* - NO_SUCH_FIELD if a field does not exist in an object
|
||||||
|
* - INDEX_OUT_OF_BOUNDS if an array index is larger than an array length
|
||||||
|
* - INCORRECT_TYPE if a non-integer is used to access an array
|
||||||
|
*/
|
||||||
|
inline simdjson_result<element> at_path(std::string_view json_path) const noexcept;
|
||||||
|
|
||||||
#ifndef SIMDJSON_DISABLE_DEPRECATED_API
|
#ifndef SIMDJSON_DISABLE_DEPRECATED_API
|
||||||
/**
|
/**
|
||||||
*
|
*
|
||||||
@@ -526,6 +541,7 @@ public:
|
|||||||
simdjson_inline simdjson_result<dom::element> operator[](std::string_view key) const noexcept;
|
simdjson_inline simdjson_result<dom::element> operator[](std::string_view key) const noexcept;
|
||||||
simdjson_inline simdjson_result<dom::element> operator[](const char *key) const noexcept;
|
simdjson_inline simdjson_result<dom::element> operator[](const char *key) const noexcept;
|
||||||
simdjson_inline simdjson_result<dom::element> at_pointer(const std::string_view json_pointer) const noexcept;
|
simdjson_inline simdjson_result<dom::element> at_pointer(const std::string_view json_pointer) const noexcept;
|
||||||
|
simdjson_inline simdjson_result<dom::element> at_path(const std::string_view json_path) const noexcept;
|
||||||
[[deprecated("For standard compliance, use at_pointer instead, and prefix your pointers with a slash '/', see RFC6901 ")]]
|
[[deprecated("For standard compliance, use at_pointer instead, and prefix your pointers with a slash '/', see RFC6901 ")]]
|
||||||
simdjson_inline simdjson_result<dom::element> at(const std::string_view json_pointer) const noexcept;
|
simdjson_inline simdjson_result<dom::element> at(const std::string_view json_pointer) const noexcept;
|
||||||
simdjson_inline simdjson_result<dom::element> at(size_t index) const noexcept;
|
simdjson_inline simdjson_result<dom::element> at(size_t index) const noexcept;
|
||||||
|
|||||||
@@ -7,6 +7,7 @@
|
|||||||
|
|
||||||
#include "simdjson/dom/element-inl.h"
|
#include "simdjson/dom/element-inl.h"
|
||||||
#include "simdjson/error-inl.h"
|
#include "simdjson/error-inl.h"
|
||||||
|
#include "simdjson/jsonpathutil.h"
|
||||||
|
|
||||||
#include <cstring>
|
#include <cstring>
|
||||||
|
|
||||||
@@ -34,6 +35,11 @@ inline simdjson_result<dom::element> simdjson_result<dom::object>::at_pointer(st
|
|||||||
if (error()) { return error(); }
|
if (error()) { return error(); }
|
||||||
return first.at_pointer(json_pointer);
|
return first.at_pointer(json_pointer);
|
||||||
}
|
}
|
||||||
|
inline simdjson_result<dom::element> simdjson_result<dom::object>::at_path(std::string_view json_path) const noexcept {
|
||||||
|
auto json_pointer = json_path_to_pointer_conversion(json_path);
|
||||||
|
if (json_pointer == "-1") { return INVALID_JSON_POINTER; }
|
||||||
|
return at_pointer(json_pointer);
|
||||||
|
}
|
||||||
inline simdjson_result<dom::element> simdjson_result<dom::object>::at_key(std::string_view key) const noexcept {
|
inline simdjson_result<dom::element> simdjson_result<dom::object>::at_key(std::string_view key) const noexcept {
|
||||||
if (error()) { return error(); }
|
if (error()) { return error(); }
|
||||||
return first.at_key(key);
|
return first.at_key(key);
|
||||||
@@ -131,6 +137,12 @@ inline simdjson_result<element> object::at_pointer(std::string_view json_pointer
|
|||||||
return child;
|
return child;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
inline simdjson_result<element> object::at_path(std::string_view json_path) const noexcept {
|
||||||
|
auto json_pointer = json_path_to_pointer_conversion(json_path);
|
||||||
|
if (json_pointer == "-1") { return INVALID_JSON_POINTER; }
|
||||||
|
return at_pointer(json_pointer);
|
||||||
|
}
|
||||||
|
|
||||||
inline simdjson_result<element> object::at_key(std::string_view key) const noexcept {
|
inline simdjson_result<element> object::at_key(std::string_view key) const noexcept {
|
||||||
iterator end_field = end();
|
iterator end_field = end();
|
||||||
for (iterator field = begin(); field != end_field; ++field) {
|
for (iterator field = begin(); field != end_field; ++field) {
|
||||||
@@ -153,6 +165,10 @@ inline simdjson_result<element> object::at_key_case_insensitive(std::string_view
|
|||||||
return NO_SUCH_FIELD;
|
return NO_SUCH_FIELD;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
inline object::operator element() const noexcept {
|
||||||
|
return element(tape);
|
||||||
|
}
|
||||||
|
|
||||||
//
|
//
|
||||||
// object::iterator inline implementation
|
// object::iterator inline implementation
|
||||||
//
|
//
|
||||||
|
|||||||
@@ -171,6 +171,21 @@ public:
|
|||||||
*/
|
*/
|
||||||
inline simdjson_result<element> at_pointer(std::string_view json_pointer) const noexcept;
|
inline simdjson_result<element> at_pointer(std::string_view json_pointer) const noexcept;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Get the value associated with the given JSONPath expression. We only support
|
||||||
|
* JSONPath queries that trivially convertible to JSON Pointer queries: key
|
||||||
|
* names and array indices.
|
||||||
|
*
|
||||||
|
* https://datatracker.ietf.org/doc/html/draft-normington-jsonpath-00
|
||||||
|
*
|
||||||
|
* @return The value associated with the given JSONPath expression, or:
|
||||||
|
* - INVALID_JSON_POINTER if the JSONPath to JSON Pointer conversion fails
|
||||||
|
* - NO_SUCH_FIELD if a field does not exist in an object
|
||||||
|
* - INDEX_OUT_OF_BOUNDS if an array index is larger than an array length
|
||||||
|
* - INCORRECT_TYPE if a non-integer is used to access an array
|
||||||
|
*/
|
||||||
|
inline simdjson_result<element> at_path(std::string_view json_path) const noexcept;
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Get the value associated with the given key.
|
* Get the value associated with the given key.
|
||||||
*
|
*
|
||||||
@@ -200,6 +215,11 @@ public:
|
|||||||
*/
|
*/
|
||||||
inline simdjson_result<element> at_key_case_insensitive(std::string_view key) const noexcept;
|
inline simdjson_result<element> at_key_case_insensitive(std::string_view key) const noexcept;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Implicitly convert object to element
|
||||||
|
*/
|
||||||
|
inline operator element() const noexcept;
|
||||||
|
|
||||||
private:
|
private:
|
||||||
simdjson_inline object(const internal::tape_ref &tape) noexcept;
|
simdjson_inline object(const internal::tape_ref &tape) noexcept;
|
||||||
|
|
||||||
@@ -239,6 +259,7 @@ public:
|
|||||||
inline simdjson_result<dom::element> operator[](std::string_view key) const noexcept;
|
inline simdjson_result<dom::element> operator[](std::string_view key) const noexcept;
|
||||||
inline simdjson_result<dom::element> operator[](const char *key) const noexcept;
|
inline simdjson_result<dom::element> operator[](const char *key) const noexcept;
|
||||||
inline simdjson_result<dom::element> at_pointer(std::string_view json_pointer) const noexcept;
|
inline simdjson_result<dom::element> at_pointer(std::string_view json_pointer) const noexcept;
|
||||||
|
inline simdjson_result<dom::element> at_path(std::string_view json_path) const noexcept;
|
||||||
inline simdjson_result<dom::element> at_key(std::string_view key) const noexcept;
|
inline simdjson_result<dom::element> at_key(std::string_view key) const noexcept;
|
||||||
inline simdjson_result<dom::element> at_key_case_insensitive(std::string_view key) const noexcept;
|
inline simdjson_result<dom::element> at_key_case_insensitive(std::string_view key) const noexcept;
|
||||||
|
|
||||||
|
|||||||
@@ -191,7 +191,7 @@ simdjson_inline size_t parser::capacity() const noexcept {
|
|||||||
simdjson_inline size_t parser::max_capacity() const noexcept {
|
simdjson_inline size_t parser::max_capacity() const noexcept {
|
||||||
return _max_capacity;
|
return _max_capacity;
|
||||||
}
|
}
|
||||||
simdjson_inline size_t parser::max_depth() const noexcept {
|
simdjson_pure simdjson_inline size_t parser::max_depth() const noexcept {
|
||||||
return implementation ? implementation->max_depth() : DEFAULT_MAX_DEPTH;
|
return implementation ? implementation->max_depth() : DEFAULT_MAX_DEPTH;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -527,7 +527,7 @@ public:
|
|||||||
*
|
*
|
||||||
* @return Maximum depth, in bytes.
|
* @return Maximum depth, in bytes.
|
||||||
*/
|
*/
|
||||||
simdjson_inline size_t max_depth() const noexcept;
|
simdjson_pure simdjson_inline size_t max_depth() const noexcept;
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Set max_capacity. This is the largest document this parser can automatically support.
|
* Set max_capacity. This is the largest document this parser can automatically support.
|
||||||
@@ -549,9 +549,14 @@ public:
|
|||||||
/**
|
/**
|
||||||
* The parser instance can use threads when they are available to speed up some
|
* The parser instance can use threads when they are available to speed up some
|
||||||
* operations. It is enabled by default. Changing this attribute will change the
|
* operations. It is enabled by default. Changing this attribute will change the
|
||||||
* behavior of the parser for future operations.
|
* behavior of the parser for future operations. Set to true by default.
|
||||||
*/
|
*/
|
||||||
bool threaded{true};
|
bool threaded{true};
|
||||||
|
#else
|
||||||
|
/**
|
||||||
|
* When SIMDJSON_THREADS_ENABLED is not defined, the parser instance cannot use threads.
|
||||||
|
*/
|
||||||
|
bool threaded{false};
|
||||||
#endif
|
#endif
|
||||||
/** @private Use the new DOM API instead */
|
/** @private Use the new DOM API instead */
|
||||||
class Iterator;
|
class Iterator;
|
||||||
|
|||||||
@@ -57,15 +57,15 @@ public:
|
|||||||
simdjson_inline void one_char(char c);
|
simdjson_inline void one_char(char c);
|
||||||
|
|
||||||
simdjson_inline void call_print_newline() {
|
simdjson_inline void call_print_newline() {
|
||||||
this->print_newline();
|
static_cast<formatter*>(this)->print_newline();
|
||||||
}
|
}
|
||||||
|
|
||||||
simdjson_inline void call_print_indents(size_t depth) {
|
simdjson_inline void call_print_indents(size_t depth) {
|
||||||
this->print_indents(depth);
|
static_cast<formatter*>(this)->print_indents(depth);
|
||||||
}
|
}
|
||||||
|
|
||||||
simdjson_inline void call_print_space() {
|
simdjson_inline void call_print_space() {
|
||||||
this->print_space();
|
static_cast<formatter*>(this)->print_space();
|
||||||
}
|
}
|
||||||
|
|
||||||
protected:
|
protected:
|
||||||
|
|||||||
@@ -39,7 +39,7 @@ enum error_code {
|
|||||||
INDEX_OUT_OF_BOUNDS, ///< JSON array index too large
|
INDEX_OUT_OF_BOUNDS, ///< JSON array index too large
|
||||||
NO_SUCH_FIELD, ///< JSON field not found in object
|
NO_SUCH_FIELD, ///< JSON field not found in object
|
||||||
IO_ERROR, ///< Error reading a file
|
IO_ERROR, ///< Error reading a file
|
||||||
INVALID_JSON_POINTER, ///< Invalid JSON pointer reference
|
INVALID_JSON_POINTER, ///< Invalid JSON pointer syntax
|
||||||
INVALID_URI_FRAGMENT, ///< Invalid URI fragment
|
INVALID_URI_FRAGMENT, ///< Invalid URI fragment
|
||||||
UNEXPECTED_ERROR, ///< indicative of a bug in simdjson
|
UNEXPECTED_ERROR, ///< indicative of a bug in simdjson
|
||||||
PARSER_IN_USE, ///< parser is already in use.
|
PARSER_IN_USE, ///< parser is already in use.
|
||||||
|
|||||||
@@ -56,13 +56,13 @@ static simdjson_inline uint64_t _umul128(uint64_t ab, uint64_t cd, uint64_t *hi)
|
|||||||
simdjson_inline internal::value128 full_multiplication(uint64_t value1, uint64_t value2) {
|
simdjson_inline internal::value128 full_multiplication(uint64_t value1, uint64_t value2) {
|
||||||
internal::value128 answer;
|
internal::value128 answer;
|
||||||
#if SIMDJSON_REGULAR_VISUAL_STUDIO || SIMDJSON_IS_32BITS
|
#if SIMDJSON_REGULAR_VISUAL_STUDIO || SIMDJSON_IS_32BITS
|
||||||
#ifdef _M_ARM64
|
#if SIMDJSON_IS_ARM64
|
||||||
// ARM64 has native support for 64-bit multiplications, no need to emultate
|
// ARM64 has native support for 64-bit multiplications, no need to emultate
|
||||||
answer.high = __umulh(value1, value2);
|
answer.high = __umulh(value1, value2);
|
||||||
answer.low = value1 * value2;
|
answer.low = value1 * value2;
|
||||||
#else
|
#else
|
||||||
answer.low = _umul128(value1, value2, &answer.high); // _umul128 not available on ARM64
|
answer.low = _umul128(value1, value2, &answer.high); // _umul128 not available on ARM64
|
||||||
#endif // _M_ARM64
|
#endif // SIMDJSON_IS_ARM64
|
||||||
#else // SIMDJSON_REGULAR_VISUAL_STUDIO || SIMDJSON_IS_32BITS
|
#else // SIMDJSON_REGULAR_VISUAL_STUDIO || SIMDJSON_IS_32BITS
|
||||||
__uint128_t r = (static_cast<__uint128_t>(value1)) * value2;
|
__uint128_t r = (static_cast<__uint128_t>(value1)) * value2;
|
||||||
answer.low = uint64_t(r);
|
answer.low = uint64_t(r);
|
||||||
@@ -75,6 +75,12 @@ simdjson_inline internal::value128 full_multiplication(uint64_t value1, uint64_t
|
|||||||
} // namespace fallback
|
} // namespace fallback
|
||||||
} // namespace simdjson
|
} // namespace simdjson
|
||||||
|
|
||||||
|
#ifndef SIMDJSON_SWAR_NUMBER_PARSING
|
||||||
|
#if SIMDJSON_IS_BIG_ENDIAN
|
||||||
|
#define SIMDJSON_SWAR_NUMBER_PARSING 0
|
||||||
|
#else
|
||||||
#define SIMDJSON_SWAR_NUMBER_PARSING 1
|
#define SIMDJSON_SWAR_NUMBER_PARSING 1
|
||||||
|
#endif
|
||||||
|
#endif
|
||||||
|
|
||||||
#endif // SIMDJSON_FALLBACK_NUMBERPARSING_DEFS_H
|
#endif // SIMDJSON_FALLBACK_NUMBERPARSING_DEFS_H
|
||||||
|
|||||||
@@ -574,7 +574,6 @@ simdjson_unused simdjson_inline simdjson_result<number_type> get_number_type(con
|
|||||||
// Our objective is accurate parsing (ULP of 0) at high speed.
|
// Our objective is accurate parsing (ULP of 0) at high speed.
|
||||||
template<typename W>
|
template<typename W>
|
||||||
simdjson_inline error_code parse_number(const uint8_t *const src, W &writer) {
|
simdjson_inline error_code parse_number(const uint8_t *const src, W &writer) {
|
||||||
|
|
||||||
//
|
//
|
||||||
// Check for minus sign
|
// Check for minus sign
|
||||||
//
|
//
|
||||||
|
|||||||
@@ -2,6 +2,7 @@
|
|||||||
|
|
||||||
#ifndef SIMDJSON_CONDITIONAL_INCLUDE
|
#ifndef SIMDJSON_CONDITIONAL_INCLUDE
|
||||||
#define SIMDJSON_GENERIC_ONDEMAND_ARRAY_INL_H
|
#define SIMDJSON_GENERIC_ONDEMAND_ARRAY_INL_H
|
||||||
|
#include "simdjson/jsonpathutil.h"
|
||||||
#include "simdjson/generic/ondemand/base.h"
|
#include "simdjson/generic/ondemand/base.h"
|
||||||
#include "simdjson/generic/ondemand/array.h"
|
#include "simdjson/generic/ondemand/array.h"
|
||||||
#include "simdjson/generic/ondemand/array_iterator-inl.h"
|
#include "simdjson/generic/ondemand/array_iterator-inl.h"
|
||||||
@@ -163,53 +164,6 @@ inline simdjson_result<value> array::at_pointer(std::string_view json_pointer) n
|
|||||||
return child;
|
return child;
|
||||||
}
|
}
|
||||||
|
|
||||||
inline std::string json_path_to_pointer_conversion(std::string_view json_path) {
|
|
||||||
if (json_path.empty() || (json_path.front() != '.' &&
|
|
||||||
json_path.front() != '[')) {
|
|
||||||
return "-1"; // This is just a sentinel value, the caller should check for this and return an error.
|
|
||||||
}
|
|
||||||
|
|
||||||
std::string result;
|
|
||||||
// Reserve space to reduce allocations, adjusting for potential increases due
|
|
||||||
// to escaping.
|
|
||||||
result.reserve(json_path.size() * 2);
|
|
||||||
|
|
||||||
size_t i = 0;
|
|
||||||
|
|
||||||
while (i < json_path.length()) {
|
|
||||||
if (json_path[i] == '.') {
|
|
||||||
result += '/';
|
|
||||||
} else if (json_path[i] == '[') {
|
|
||||||
result += '/';
|
|
||||||
++i; // Move past the '['
|
|
||||||
while (i < json_path.length() && json_path[i] != ']') {
|
|
||||||
if (json_path[i] == '~') {
|
|
||||||
result += "~0";
|
|
||||||
} else if (json_path[i] == '/') {
|
|
||||||
result += "~1";
|
|
||||||
} else {
|
|
||||||
result += json_path[i];
|
|
||||||
}
|
|
||||||
++i;
|
|
||||||
}
|
|
||||||
if (i == json_path.length() || json_path[i] != ']') {
|
|
||||||
return "-1"; // Using sentinel value that will be handled as an error by the caller.
|
|
||||||
}
|
|
||||||
} else {
|
|
||||||
if (json_path[i] == '~') {
|
|
||||||
result += "~0";
|
|
||||||
} else if (json_path[i] == '/') {
|
|
||||||
result += "~1";
|
|
||||||
} else {
|
|
||||||
result += json_path[i];
|
|
||||||
}
|
|
||||||
}
|
|
||||||
++i;
|
|
||||||
}
|
|
||||||
|
|
||||||
return result;
|
|
||||||
}
|
|
||||||
|
|
||||||
inline simdjson_result<value> array::at_path(std::string_view json_path) noexcept {
|
inline simdjson_result<value> array::at_path(std::string_view json_path) noexcept {
|
||||||
auto json_pointer = json_path_to_pointer_conversion(json_path);
|
auto json_pointer = json_path_to_pointer_conversion(json_path);
|
||||||
if (json_pointer == "-1") { return INVALID_JSON_POINTER; }
|
if (json_pointer == "-1") { return INVALID_JSON_POINTER; }
|
||||||
|
|||||||
@@ -13,5 +13,6 @@
|
|||||||
#include "simdjson/padded_string.h"
|
#include "simdjson/padded_string.h"
|
||||||
#include "simdjson/padded_string_view.h"
|
#include "simdjson/padded_string_view.h"
|
||||||
#include "simdjson/internal/dom_parser_implementation.h"
|
#include "simdjson/internal/dom_parser_implementation.h"
|
||||||
|
#include "simdjson/jsonpathutil.h"
|
||||||
|
|
||||||
#endif // SIMDJSON_GENERIC_ONDEMAND_DEPENDENCIES_H
|
#endif // SIMDJSON_GENERIC_ONDEMAND_DEPENDENCIES_H
|
||||||
@@ -3,16 +3,14 @@
|
|||||||
#ifndef SIMDJSON_CONDITIONAL_INCLUDE
|
#ifndef SIMDJSON_CONDITIONAL_INCLUDE
|
||||||
#define SIMDJSON_GENERIC_ONDEMAND_DOCUMENT_INL_H
|
#define SIMDJSON_GENERIC_ONDEMAND_DOCUMENT_INL_H
|
||||||
#include "simdjson/generic/ondemand/base.h"
|
#include "simdjson/generic/ondemand/base.h"
|
||||||
#include "simdjson/generic/ondemand/array-inl.h"
|
|
||||||
#include "simdjson/generic/ondemand/array_iterator.h"
|
#include "simdjson/generic/ondemand/array_iterator.h"
|
||||||
#include "simdjson/generic/ondemand/document.h"
|
#include "simdjson/generic/ondemand/document.h"
|
||||||
#include "simdjson/generic/ondemand/json_iterator-inl.h"
|
|
||||||
#include "simdjson/generic/ondemand/json_path_to_pointer_conversion.h"
|
|
||||||
#include "simdjson/generic/ondemand/json_path_to_pointer_conversion-inl.h"
|
|
||||||
#include "simdjson/generic/ondemand/json_type.h"
|
#include "simdjson/generic/ondemand/json_type.h"
|
||||||
#include "simdjson/generic/ondemand/object-inl.h"
|
|
||||||
#include "simdjson/generic/ondemand/raw_json_string.h"
|
#include "simdjson/generic/ondemand/raw_json_string.h"
|
||||||
#include "simdjson/generic/ondemand/value.h"
|
#include "simdjson/generic/ondemand/value.h"
|
||||||
|
#include "simdjson/generic/ondemand/array-inl.h"
|
||||||
|
#include "simdjson/generic/ondemand/json_iterator-inl.h"
|
||||||
|
#include "simdjson/generic/ondemand/object-inl.h"
|
||||||
#include "simdjson/generic/ondemand/value_iterator-inl.h"
|
#include "simdjson/generic/ondemand/value_iterator-inl.h"
|
||||||
#endif // SIMDJSON_CONDITIONAL_INCLUDE
|
#endif // SIMDJSON_CONDITIONAL_INCLUDE
|
||||||
|
|
||||||
@@ -167,24 +165,26 @@ template<> simdjson_inline simdjson_result<int64_t> document::get() & noexcept {
|
|||||||
template<> simdjson_inline simdjson_result<bool> document::get() & noexcept { return get_bool(); }
|
template<> simdjson_inline simdjson_result<bool> document::get() & noexcept { return get_bool(); }
|
||||||
template<> simdjson_inline simdjson_result<value> document::get() & noexcept { return get_value(); }
|
template<> simdjson_inline simdjson_result<value> document::get() & noexcept { return get_value(); }
|
||||||
|
|
||||||
template<> simdjson_inline simdjson_result<raw_json_string> document::get() && noexcept { return get_raw_json_string(); }
|
template<> simdjson_deprecated simdjson_inline simdjson_result<raw_json_string> document::get() && noexcept { return get_raw_json_string(); }
|
||||||
template<> simdjson_inline simdjson_result<std::string_view> document::get() && noexcept { return get_string(false); }
|
template<> simdjson_deprecated simdjson_inline simdjson_result<std::string_view> document::get() && noexcept { return get_string(false); }
|
||||||
template<> simdjson_inline simdjson_result<double> document::get() && noexcept { return std::forward<document>(*this).get_double(); }
|
template<> simdjson_deprecated simdjson_inline simdjson_result<double> document::get() && noexcept { return std::forward<document>(*this).get_double(); }
|
||||||
template<> simdjson_inline simdjson_result<uint64_t> document::get() && noexcept { return std::forward<document>(*this).get_uint64(); }
|
template<> simdjson_deprecated simdjson_inline simdjson_result<uint64_t> document::get() && noexcept { return std::forward<document>(*this).get_uint64(); }
|
||||||
template<> simdjson_inline simdjson_result<int64_t> document::get() && noexcept { return std::forward<document>(*this).get_int64(); }
|
template<> simdjson_deprecated simdjson_inline simdjson_result<int64_t> document::get() && noexcept { return std::forward<document>(*this).get_int64(); }
|
||||||
template<> simdjson_inline simdjson_result<bool> document::get() && noexcept { return std::forward<document>(*this).get_bool(); }
|
template<> simdjson_deprecated simdjson_inline simdjson_result<bool> document::get() && noexcept { return std::forward<document>(*this).get_bool(); }
|
||||||
template<> simdjson_inline simdjson_result<value> document::get() && noexcept { return get_value(); }
|
template<> simdjson_deprecated simdjson_inline simdjson_result<value> document::get() && noexcept { return get_value(); }
|
||||||
|
|
||||||
template<typename T> simdjson_inline error_code document::get(T &out) & noexcept {
|
template<typename T> simdjson_inline error_code document::get(T &out) & noexcept {
|
||||||
return get<T>().get(out);
|
return get<T>().get(out);
|
||||||
}
|
}
|
||||||
template<typename T> simdjson_inline error_code document::get(T &out) && noexcept {
|
template<typename T> simdjson_deprecated simdjson_inline error_code document::get(T &out) && noexcept {
|
||||||
return std::forward<document>(*this).get<T>().get(out);
|
return std::forward<document>(*this).get<T>().get(out);
|
||||||
}
|
}
|
||||||
|
|
||||||
#if SIMDJSON_EXCEPTIONS
|
#if SIMDJSON_EXCEPTIONS
|
||||||
template <class T>
|
template <class T>
|
||||||
simdjson_inline document::operator T() noexcept(false) { return get<T>(); }
|
simdjson_deprecated simdjson_inline document::operator T() && noexcept(false) { return get<T>(); }
|
||||||
|
template <class T>
|
||||||
|
simdjson_inline document::operator T() & noexcept(false) { return get<T>(); }
|
||||||
simdjson_inline document::operator array() & noexcept(false) { return get_array(); }
|
simdjson_inline document::operator array() & noexcept(false) { return get_array(); }
|
||||||
simdjson_inline document::operator object() & noexcept(false) { return get_object(); }
|
simdjson_inline document::operator object() & noexcept(false) { return get_object(); }
|
||||||
simdjson_inline document::operator uint64_t() noexcept(false) { return get_uint64(); }
|
simdjson_inline document::operator uint64_t() noexcept(false) { return get_uint64(); }
|
||||||
@@ -471,7 +471,7 @@ simdjson_inline simdjson_result<T> simdjson_result<SIMDJSON_IMPLEMENTATION::onde
|
|||||||
return first.get<T>();
|
return first.get<T>();
|
||||||
}
|
}
|
||||||
template<typename T>
|
template<typename T>
|
||||||
simdjson_inline simdjson_result<T> simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::document>::get() && noexcept {
|
simdjson_deprecated simdjson_inline simdjson_result<T> simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::document>::get() && noexcept {
|
||||||
if (error()) { return error(); }
|
if (error()) { return error(); }
|
||||||
return std::forward<SIMDJSON_IMPLEMENTATION::ondemand::document>(first).get<T>();
|
return std::forward<SIMDJSON_IMPLEMENTATION::ondemand::document>(first).get<T>();
|
||||||
}
|
}
|
||||||
@@ -487,7 +487,7 @@ simdjson_inline error_code simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::do
|
|||||||
}
|
}
|
||||||
|
|
||||||
template<> simdjson_inline simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::document> simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::document>::get<SIMDJSON_IMPLEMENTATION::ondemand::document>() & noexcept = delete;
|
template<> simdjson_inline simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::document> simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::document>::get<SIMDJSON_IMPLEMENTATION::ondemand::document>() & noexcept = delete;
|
||||||
template<> simdjson_inline simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::document> simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::document>::get<SIMDJSON_IMPLEMENTATION::ondemand::document>() && noexcept {
|
template<> simdjson_deprecated simdjson_inline simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::document> simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::document>::get<SIMDJSON_IMPLEMENTATION::ondemand::document>() && noexcept {
|
||||||
if (error()) { return error(); }
|
if (error()) { return error(); }
|
||||||
return std::forward<SIMDJSON_IMPLEMENTATION::ondemand::document>(first);
|
return std::forward<SIMDJSON_IMPLEMENTATION::ondemand::document>(first);
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -188,7 +188,7 @@ public:
|
|||||||
" You may also add support for custom types, see our documentation.");
|
" You may also add support for custom types, see our documentation.");
|
||||||
}
|
}
|
||||||
/** @overload template<typename T> simdjson_result<T> get() & noexcept */
|
/** @overload template<typename T> simdjson_result<T> get() & noexcept */
|
||||||
template<typename T> simdjson_inline simdjson_result<T> get() && noexcept {
|
template<typename T> simdjson_deprecated simdjson_inline simdjson_result<T> get() && noexcept {
|
||||||
// Unless the simdjson library or the user provides an inline implementation, calling this method should
|
// Unless the simdjson library or the user provides an inline implementation, calling this method should
|
||||||
// immediately fail.
|
// immediately fail.
|
||||||
static_assert(!sizeof(T), "The get method with given type is not implemented by the simdjson library. "
|
static_assert(!sizeof(T), "The get method with given type is not implemented by the simdjson library. "
|
||||||
@@ -211,7 +211,7 @@ public:
|
|||||||
*/
|
*/
|
||||||
template<typename T> simdjson_inline error_code get(T &out) & noexcept;
|
template<typename T> simdjson_inline error_code get(T &out) & noexcept;
|
||||||
/** @overload template<typename T> error_code get(T &out) & noexcept */
|
/** @overload template<typename T> error_code get(T &out) & noexcept */
|
||||||
template<typename T> simdjson_inline error_code get(T &out) && noexcept;
|
template<typename T> simdjson_deprecated simdjson_inline error_code get(T &out) && noexcept;
|
||||||
|
|
||||||
#if SIMDJSON_EXCEPTIONS
|
#if SIMDJSON_EXCEPTIONS
|
||||||
/**
|
/**
|
||||||
@@ -224,7 +224,10 @@ public:
|
|||||||
* @returns An instance of type T
|
* @returns An instance of type T
|
||||||
*/
|
*/
|
||||||
template <class T>
|
template <class T>
|
||||||
explicit simdjson_inline operator T() noexcept(false);
|
explicit simdjson_inline operator T() & noexcept(false);
|
||||||
|
template <class T>
|
||||||
|
explicit simdjson_deprecated simdjson_inline operator T() && noexcept(false);
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Cast this JSON value to an array.
|
* Cast this JSON value to an array.
|
||||||
*
|
*
|
||||||
@@ -688,6 +691,11 @@ protected:
|
|||||||
|
|
||||||
/**
|
/**
|
||||||
* A document_reference is a thin wrapper around a document reference instance.
|
* A document_reference is a thin wrapper around a document reference instance.
|
||||||
|
* The document_reference instances are used primarily/solely for streams of JSON
|
||||||
|
* documents. They differ from document instances when parsing a scalar value
|
||||||
|
* (a document that is not an array or an object). In the case of a document,
|
||||||
|
* we expect the document to be fully consumed. In the case of a document_reference,
|
||||||
|
* we allow trailing content.
|
||||||
*/
|
*/
|
||||||
class document_reference {
|
class document_reference {
|
||||||
public:
|
public:
|
||||||
@@ -790,7 +798,7 @@ public:
|
|||||||
simdjson_inline simdjson_result<bool> is_null() noexcept;
|
simdjson_inline simdjson_result<bool> is_null() noexcept;
|
||||||
|
|
||||||
template<typename T> simdjson_inline simdjson_result<T> get() & noexcept;
|
template<typename T> simdjson_inline simdjson_result<T> get() & noexcept;
|
||||||
template<typename T> simdjson_inline simdjson_result<T> get() && noexcept;
|
template<typename T> simdjson_deprecated simdjson_inline simdjson_result<T> get() && noexcept;
|
||||||
|
|
||||||
template<typename T> simdjson_inline error_code get(T &out) & noexcept;
|
template<typename T> simdjson_inline error_code get(T &out) & noexcept;
|
||||||
template<typename T> simdjson_inline error_code get(T &out) && noexcept;
|
template<typename T> simdjson_inline error_code get(T &out) && noexcept;
|
||||||
|
|||||||
@@ -342,9 +342,18 @@ simdjson_inline std::string_view document_stream::iterator::source() const noexc
|
|||||||
depth--;
|
depth--;
|
||||||
break;
|
break;
|
||||||
default: // Scalar value document
|
default: // Scalar value document
|
||||||
// TODO: Remove any trailing whitespaces
|
// TODO: We could remove trailing whitespaces
|
||||||
// This returns a string spanning from start of value to the beginning of the next document (excluded)
|
// This returns a string spanning from start of value to the beginning of the next document (excluded)
|
||||||
return std::string_view(reinterpret_cast<const char*>(stream->buf) + current_index(), stream->parser->implementation->structural_indexes[++cur_struct_index] - current_index() - 1);
|
{
|
||||||
|
auto next_index = stream->parser->implementation->structural_indexes[++cur_struct_index];
|
||||||
|
// normally the length would be next_index - current_index() - 1, except for the last document
|
||||||
|
size_t svlen = next_index - current_index();
|
||||||
|
const char *start = reinterpret_cast<const char*>(stream->buf) + current_index();
|
||||||
|
while(svlen > 1 && (std::isspace(start[svlen-1]) || start[svlen-1] == '\0')) {
|
||||||
|
svlen--;
|
||||||
|
}
|
||||||
|
return std::string_view(start, svlen);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
cur_struct_index++;
|
cur_struct_index++;
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -316,7 +316,7 @@ private:
|
|||||||
friend class document;
|
friend class document;
|
||||||
friend class json_iterator;
|
friend class json_iterator;
|
||||||
friend struct simdjson_result<ondemand::document_stream>;
|
friend struct simdjson_result<ondemand::document_stream>;
|
||||||
friend struct internal::simdjson_result_base<ondemand::document_stream>;
|
friend struct simdjson::internal::simdjson_result_base<ondemand::document_stream>;
|
||||||
}; // document_stream
|
}; // document_stream
|
||||||
|
|
||||||
} // namespace ondemand
|
} // namespace ondemand
|
||||||
|
|||||||
@@ -38,6 +38,14 @@ simdjson_inline simdjson_warn_unused simdjson_result<std::string_view> field::un
|
|||||||
return answer;
|
return answer;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
template <typename string_type>
|
||||||
|
simdjson_inline simdjson_warn_unused error_code field::unescaped_key(string_type& receiver, bool allow_replacement) noexcept {
|
||||||
|
std::string_view key;
|
||||||
|
SIMDJSON_TRY( unescaped_key(allow_replacement).get(key) );
|
||||||
|
receiver = key;
|
||||||
|
return SUCCESS;
|
||||||
|
}
|
||||||
|
|
||||||
simdjson_inline raw_json_string field::key() const noexcept {
|
simdjson_inline raw_json_string field::key() const noexcept {
|
||||||
SIMDJSON_ASSUME(first.buf != nullptr); // We would like to call .alive() by Visual Studio won't let us.
|
SIMDJSON_ASSUME(first.buf != nullptr); // We would like to call .alive() by Visual Studio won't let us.
|
||||||
return first;
|
return first;
|
||||||
@@ -105,6 +113,12 @@ simdjson_inline simdjson_result<std::string_view> simdjson_result<SIMDJSON_IMPLE
|
|||||||
return first.unescaped_key(allow_replacement);
|
return first.unescaped_key(allow_replacement);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
template<typename string_type>
|
||||||
|
simdjson_inline error_code simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::field>::unescaped_key(string_type &receiver, bool allow_replacement) noexcept {
|
||||||
|
if (error()) { return error(); }
|
||||||
|
return first.unescaped_key(receiver, allow_replacement);
|
||||||
|
}
|
||||||
|
|
||||||
simdjson_inline simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::value> simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::field>::value() noexcept {
|
simdjson_inline simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::value> simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::field>::value() noexcept {
|
||||||
if (error()) { return error(); }
|
if (error()) { return error(); }
|
||||||
return std::move(first.value());
|
return std::move(first.value());
|
||||||
|
|||||||
@@ -36,21 +36,37 @@ public:
|
|||||||
* This consumes the key: once you have called unescaped_key(), you cannot
|
* This consumes the key: once you have called unescaped_key(), you cannot
|
||||||
* call it again nor can you call key().
|
* call it again nor can you call key().
|
||||||
*/
|
*/
|
||||||
simdjson_inline simdjson_warn_unused simdjson_result<std::string_view> unescaped_key(bool allow_replacement) noexcept;
|
simdjson_inline simdjson_warn_unused simdjson_result<std::string_view> unescaped_key(bool allow_replacement = false) noexcept;
|
||||||
|
/**
|
||||||
|
* Get the key as a string_view (for higher speed, consider raw_key).
|
||||||
|
* We deliberately use a more cumbersome name (unescaped_key) to force users
|
||||||
|
* to think twice about using it. The content is stored in the receiver.
|
||||||
|
*
|
||||||
|
* This consumes the key: once you have called unescaped_key(), you cannot
|
||||||
|
* call it again nor can you call key().
|
||||||
|
*/
|
||||||
|
template <typename string_type>
|
||||||
|
simdjson_inline simdjson_warn_unused error_code unescaped_key(string_type& receiver, bool allow_replacement = false) noexcept;
|
||||||
/**
|
/**
|
||||||
* Get the key as a raw_json_string. Can be used for direct comparison with
|
* Get the key as a raw_json_string. Can be used for direct comparison with
|
||||||
* an unescaped C string: e.g., key() == "test".
|
* an unescaped C string: e.g., key() == "test". This does not count as
|
||||||
|
* consumption of the content: you can safely call it repeatedly.
|
||||||
|
* See escaped_key() for a similar function which returns
|
||||||
|
* a more convenient std::string_view result.
|
||||||
*/
|
*/
|
||||||
simdjson_inline raw_json_string key() const noexcept;
|
simdjson_inline raw_json_string key() const noexcept;
|
||||||
/**
|
/**
|
||||||
* Get the unprocessed key as a string_view. This includes the quotes and may include
|
* Get the unprocessed key as a string_view. This includes the quotes and may include
|
||||||
* some spaces after the last quote.
|
* some spaces after the last quote. This does not count as
|
||||||
|
* consumption of the content: you can safely call it repeatedly.
|
||||||
|
* See escaped_key().
|
||||||
*/
|
*/
|
||||||
simdjson_inline std::string_view key_raw_json_token() const noexcept;
|
simdjson_inline std::string_view key_raw_json_token() const noexcept;
|
||||||
/**
|
/**
|
||||||
* Get the key as a string_view. This does not include the quotes and
|
* Get the key as a string_view. This does not include the quotes and
|
||||||
* the string is unprocessed key so it may contain escape characters
|
* the string is unprocessed key so it may contain escape characters
|
||||||
* (e.g., \uXXXX or \n). Use unescaped_key() to get the unescaped key.
|
* (e.g., \uXXXX or \n). It does not count as a consumption of the content:
|
||||||
|
* you can safely call it repeatedly. Use unescaped_key() to get the unescaped key.
|
||||||
*/
|
*/
|
||||||
simdjson_inline std::string_view escaped_key() const noexcept;
|
simdjson_inline std::string_view escaped_key() const noexcept;
|
||||||
/**
|
/**
|
||||||
@@ -84,6 +100,8 @@ public:
|
|||||||
simdjson_inline simdjson_result() noexcept = default;
|
simdjson_inline simdjson_result() noexcept = default;
|
||||||
|
|
||||||
simdjson_inline simdjson_result<std::string_view> unescaped_key(bool allow_replacement = false) noexcept;
|
simdjson_inline simdjson_result<std::string_view> unescaped_key(bool allow_replacement = false) noexcept;
|
||||||
|
template<typename string_type>
|
||||||
|
simdjson_inline error_code unescaped_key(string_type &receiver, bool allow_replacement = false) noexcept;
|
||||||
simdjson_inline simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::raw_json_string> key() noexcept;
|
simdjson_inline simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::raw_json_string> key() noexcept;
|
||||||
simdjson_inline simdjson_result<std::string_view> key_raw_json_token() noexcept;
|
simdjson_inline simdjson_result<std::string_view> key_raw_json_token() noexcept;
|
||||||
simdjson_inline simdjson_result<std::string_view> escaped_key() noexcept;
|
simdjson_inline simdjson_result<std::string_view> escaped_key() noexcept;
|
||||||
|
|||||||
@@ -54,6 +54,23 @@ simdjson_inline json_iterator::json_iterator(const uint8_t *buf, ondemand::parse
|
|||||||
#endif
|
#endif
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
|
||||||
|
simdjson_inline json_iterator::json_iterator(const uint8_t *buf, ondemand::parser *_parser, bool streaming) noexcept
|
||||||
|
: token(buf, &_parser->implementation->structural_indexes[0]),
|
||||||
|
parser{_parser},
|
||||||
|
_string_buf_loc{parser->string_buf.get()},
|
||||||
|
_depth{1},
|
||||||
|
_root{parser->implementation->structural_indexes.get()},
|
||||||
|
_streaming{streaming}
|
||||||
|
|
||||||
|
{
|
||||||
|
logger::log_headers();
|
||||||
|
#if SIMDJSON_CHECK_EOF
|
||||||
|
assert_more_tokens();
|
||||||
|
#endif
|
||||||
|
}
|
||||||
|
#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
|
||||||
|
|
||||||
inline void json_iterator::rewind() noexcept {
|
inline void json_iterator::rewind() noexcept {
|
||||||
token.set_position( root_position() );
|
token.set_position( root_position() );
|
||||||
logger::log_headers(); // We start again
|
logger::log_headers(); // We start again
|
||||||
@@ -337,11 +354,23 @@ simdjson_inline token_position json_iterator::position() const noexcept {
|
|||||||
}
|
}
|
||||||
|
|
||||||
simdjson_inline simdjson_result<std::string_view> json_iterator::unescape(raw_json_string in, bool allow_replacement) noexcept {
|
simdjson_inline simdjson_result<std::string_view> json_iterator::unescape(raw_json_string in, bool allow_replacement) noexcept {
|
||||||
|
#if SIMDJSON_DEVELOPMENT_CHECKS
|
||||||
|
auto result = parser->unescape(in, _string_buf_loc, allow_replacement);
|
||||||
|
SIMDJSON_ASSUME(!parser->string_buffer_overflow(_string_buf_loc));
|
||||||
|
return result;
|
||||||
|
#else
|
||||||
return parser->unescape(in, _string_buf_loc, allow_replacement);
|
return parser->unescape(in, _string_buf_loc, allow_replacement);
|
||||||
|
#endif
|
||||||
}
|
}
|
||||||
|
|
||||||
simdjson_inline simdjson_result<std::string_view> json_iterator::unescape_wobbly(raw_json_string in) noexcept {
|
simdjson_inline simdjson_result<std::string_view> json_iterator::unescape_wobbly(raw_json_string in) noexcept {
|
||||||
|
#if SIMDJSON_DEVELOPMENT_CHECKS
|
||||||
|
auto result = parser->unescape_wobbly(in, _string_buf_loc);
|
||||||
|
SIMDJSON_ASSUME(!parser->string_buffer_overflow(_string_buf_loc));
|
||||||
|
return result;
|
||||||
|
#else
|
||||||
return parser->unescape_wobbly(in, _string_buf_loc);
|
return parser->unescape_wobbly(in, _string_buf_loc);
|
||||||
|
#endif
|
||||||
}
|
}
|
||||||
|
|
||||||
simdjson_inline void json_iterator::reenter_child(token_position position, depth_t child_depth) noexcept {
|
simdjson_inline void json_iterator::reenter_child(token_position position, depth_t child_depth) noexcept {
|
||||||
|
|||||||
@@ -293,6 +293,9 @@ public:
|
|||||||
inline bool balanced() const noexcept;
|
inline bool balanced() const noexcept;
|
||||||
protected:
|
protected:
|
||||||
simdjson_inline json_iterator(const uint8_t *buf, ondemand::parser *parser) noexcept;
|
simdjson_inline json_iterator(const uint8_t *buf, ondemand::parser *parser) noexcept;
|
||||||
|
#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
|
||||||
|
simdjson_inline json_iterator(const uint8_t *buf, ondemand::parser *parser, bool streaming) noexcept;
|
||||||
|
#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
|
||||||
/// The last token before the end
|
/// The last token before the end
|
||||||
simdjson_inline token_position last_position() const noexcept;
|
simdjson_inline token_position last_position() const noexcept;
|
||||||
/// The token *at* the end. This points at gibberish and should only be used for comparison.
|
/// The token *at* the end. This points at gibberish and should only be used for comparison.
|
||||||
|
|||||||
@@ -1,67 +0,0 @@
|
|||||||
#pragma once
|
|
||||||
|
|
||||||
#ifndef SIMDJSON_ONDEMAND_GENERIC_JSON_PATH_TO_POINTER_CONVERSION_INL_H
|
|
||||||
#define SIMDJSON_ONDEMAND_GENERIC_JSON_PATH_TO_POINTER_CONVERSION_INL_H
|
|
||||||
|
|
||||||
#ifndef SIMDJSON_CONDITIONAL_INCLUDE
|
|
||||||
#define SIMDJSON_GENERIC_ONDEMAND_JSON_PATH_TO_POINTER_CONVERSION_INL_H
|
|
||||||
#include "simdjson/generic/ondemand/json_path_to_pointer_conversion.h"
|
|
||||||
#endif // SIMDJSON_CONDITIONAL_INCLUDE
|
|
||||||
|
|
||||||
namespace simdjson {
|
|
||||||
namespace SIMDJSON_IMPLEMENTATION {}
|
|
||||||
namespace ondemand {
|
|
||||||
|
|
||||||
simdjson_inline std::string json_path_to_pointer_conversion(std::string_view json_path) {
|
|
||||||
if (json_path.empty() || (json_path.front() != '.' && json_path.front() != '[') {
|
|
||||||
return "-1"; // Sentinel value to be handled as an error by the caller.
|
|
||||||
}
|
|
||||||
|
|
||||||
std::string result;
|
|
||||||
// Reserve space to reduce allocations, adjusting for potential increases due
|
|
||||||
// to escaping.
|
|
||||||
result.reserve(json_path.size() * 2);
|
|
||||||
|
|
||||||
// Skip the initial '.' as it's assumed every path starts with it.
|
|
||||||
size_t i = 0;
|
|
||||||
|
|
||||||
while (i < json_path.length()) {
|
|
||||||
if (json_path[i] == '.') {
|
|
||||||
result += '/';
|
|
||||||
} else if (json_path[i] == '[') {
|
|
||||||
result += '/';
|
|
||||||
++i; // Move past the '['
|
|
||||||
while (i < json_path.length() && json_path[i] != ']') {
|
|
||||||
if (json_path[i] == '~') {
|
|
||||||
result += "~0";
|
|
||||||
} else if (json_path[i] == '/') {
|
|
||||||
result += "~1";
|
|
||||||
} else {
|
|
||||||
result += json_path[i];
|
|
||||||
}
|
|
||||||
++i;
|
|
||||||
}
|
|
||||||
if (i == json_path.length() || json_path[i] != ']') {
|
|
||||||
return "-1"; // Returning sentinel value that will be handled as an error by the caller
|
|
||||||
}
|
|
||||||
} else {
|
|
||||||
if (json_path[i] == '~') {
|
|
||||||
result += "~0";
|
|
||||||
} else if (json_path[i] == '/') {
|
|
||||||
result += "~1";
|
|
||||||
} else {
|
|
||||||
result += json_path[i];
|
|
||||||
}
|
|
||||||
}
|
|
||||||
++i;
|
|
||||||
}
|
|
||||||
|
|
||||||
return simdjson_result<std::string>(result);
|
|
||||||
}
|
|
||||||
|
|
||||||
|
|
||||||
} // namespace ondemand
|
|
||||||
} // namespace SIMDJSON_IMPLEMENTATION
|
|
||||||
} // namespace simdjson
|
|
||||||
|
|
||||||
#endif // SIMDJSON_ONDEMAND_GENERIC_JSON_PATH_TO_POINTER_CONVERSION_INL_H
|
|
||||||
@@ -1,22 +0,0 @@
|
|||||||
#pragma once
|
|
||||||
|
|
||||||
#ifndef SIMDJSON_ONDEMAND_GENERIC_JSON_PATH_TO_POINTER_CONVERSION_H
|
|
||||||
#define SIMDJSON_ONDEMAND_GENERIC_JSON_PATH_TO_POINTER_CONVERSION_H
|
|
||||||
|
|
||||||
namespace simdjson {
|
|
||||||
namespace SIMDJSON_IMPLEMENTATION {
|
|
||||||
namespace internal {
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Converts JSONPath to JSON Pointer.
|
|
||||||
* @param json_path The JSONPath string to be converted.
|
|
||||||
* @return A string containing the equivalent JSON Pointer.
|
|
||||||
* @throws simdjson_error If the conversion fails.
|
|
||||||
*/
|
|
||||||
simdjson_inline std::string json_path_to_pointer_conversion(std::string_view json_path);
|
|
||||||
|
|
||||||
} // namespace internal
|
|
||||||
} // namespace SIMDJSON_IMPLEMENTATION
|
|
||||||
} // namespace simdjson
|
|
||||||
|
|
||||||
#endif // SIMDJSON_JSON_PATH_TO_POINTER_CONVERSION_H
|
|
||||||
@@ -160,8 +160,8 @@ public:
|
|||||||
/**
|
/**
|
||||||
* Reset the iterator so that we are pointing back at the
|
* Reset the iterator so that we are pointing back at the
|
||||||
* beginning of the object. You should still consume values only once even if you
|
* beginning of the object. You should still consume values only once even if you
|
||||||
* can iterate through the object more than once. If you unescape a string within
|
* can iterate through the object more than once. If you unescape a string or a key
|
||||||
* the object more than once, you have unsafe code. Note that rewinding an object
|
* within the object more than once, you have unsafe code. Note that rewinding an object
|
||||||
* means that you may need to reparse it anew: it is not a free operation.
|
* means that you may need to reparse it anew: it is not a free operation.
|
||||||
*
|
*
|
||||||
* @returns true if the object contains some elements (not empty)
|
* @returns true if the object contains some elements (not empty)
|
||||||
|
|||||||
@@ -42,6 +42,11 @@ simdjson_warn_unused simdjson_inline error_code parser::allocate(size_t new_capa
|
|||||||
_max_depth = new_max_depth;
|
_max_depth = new_max_depth;
|
||||||
return SUCCESS;
|
return SUCCESS;
|
||||||
}
|
}
|
||||||
|
#if SIMDJSON_DEVELOPMENT_CHECKS
|
||||||
|
simdjson_inline simdjson_warn_unused bool parser::string_buffer_overflow(const uint8_t *string_buf_loc) const noexcept {
|
||||||
|
return (string_buf_loc < string_buf.get()) || (size_t(string_buf_loc - string_buf.get()) >= capacity());
|
||||||
|
}
|
||||||
|
#endif
|
||||||
|
|
||||||
simdjson_warn_unused simdjson_inline simdjson_result<document> parser::iterate(padded_string_view json) & noexcept {
|
simdjson_warn_unused simdjson_inline simdjson_result<document> parser::iterate(padded_string_view json) & noexcept {
|
||||||
if (json.padding() < SIMDJSON_PADDING) { return INSUFFICIENT_PADDING; }
|
if (json.padding() < SIMDJSON_PADDING) { return INSUFFICIENT_PADDING; }
|
||||||
@@ -58,6 +63,27 @@ simdjson_warn_unused simdjson_inline simdjson_result<document> parser::iterate(p
|
|||||||
return document::start({ reinterpret_cast<const uint8_t *>(json.data()), this });
|
return document::start({ reinterpret_cast<const uint8_t *>(json.data()), this });
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
|
||||||
|
simdjson_warn_unused simdjson_inline simdjson_result<document> parser::iterate_allow_incomplete_json(padded_string_view json) & noexcept {
|
||||||
|
if (json.padding() < SIMDJSON_PADDING) { return INSUFFICIENT_PADDING; }
|
||||||
|
|
||||||
|
json.remove_utf8_bom();
|
||||||
|
|
||||||
|
// Allocate if needed
|
||||||
|
if (capacity() < json.length() || !string_buf) {
|
||||||
|
SIMDJSON_TRY( allocate(json.length(), max_depth()) );
|
||||||
|
}
|
||||||
|
|
||||||
|
// Run stage 1.
|
||||||
|
const simdjson::error_code err = implementation->stage1(reinterpret_cast<const uint8_t *>(json.data()), json.length(), stage1_mode::regular);
|
||||||
|
if (err) {
|
||||||
|
if (err != UNCLOSED_STRING)
|
||||||
|
return err;
|
||||||
|
}
|
||||||
|
return document::start({ reinterpret_cast<const uint8_t *>(json.data()), this, true });
|
||||||
|
}
|
||||||
|
#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
|
||||||
|
|
||||||
simdjson_warn_unused simdjson_inline simdjson_result<document> parser::iterate(const char *json, size_t len, size_t allocated) & noexcept {
|
simdjson_warn_unused simdjson_inline simdjson_result<document> parser::iterate(const char *json, size_t len, size_t allocated) & noexcept {
|
||||||
return iterate(padded_string_view(json, len, allocated));
|
return iterate(padded_string_view(json, len, allocated));
|
||||||
}
|
}
|
||||||
@@ -129,13 +155,13 @@ inline simdjson_result<document_stream> parser::iterate_many(const padded_string
|
|||||||
return iterate_many(s.data(), s.length(), batch_size, allow_comma_separated);
|
return iterate_many(s.data(), s.length(), batch_size, allow_comma_separated);
|
||||||
}
|
}
|
||||||
|
|
||||||
simdjson_inline size_t parser::capacity() const noexcept {
|
simdjson_pure simdjson_inline size_t parser::capacity() const noexcept {
|
||||||
return _capacity;
|
return _capacity;
|
||||||
}
|
}
|
||||||
simdjson_inline size_t parser::max_capacity() const noexcept {
|
simdjson_pure simdjson_inline size_t parser::max_capacity() const noexcept {
|
||||||
return _max_capacity;
|
return _max_capacity;
|
||||||
}
|
}
|
||||||
simdjson_inline size_t parser::max_depth() const noexcept {
|
simdjson_pure simdjson_inline size_t parser::max_depth() const noexcept {
|
||||||
return _max_depth;
|
return _max_depth;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -13,8 +13,8 @@ namespace SIMDJSON_IMPLEMENTATION {
|
|||||||
namespace ondemand {
|
namespace ondemand {
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* The default batch size for document_stream instances for this On Demand kernel.
|
* The default batch size for document_stream instances for this On-Demand kernel.
|
||||||
* Note that different On Demand kernel may use a different DEFAULT_BATCH_SIZE value
|
* Note that different On-Demand kernel may use a different DEFAULT_BATCH_SIZE value
|
||||||
* in the future.
|
* in the future.
|
||||||
*/
|
*/
|
||||||
static constexpr size_t DEFAULT_BATCH_SIZE = 1000000;
|
static constexpr size_t DEFAULT_BATCH_SIZE = 1000000;
|
||||||
@@ -98,6 +98,9 @@ public:
|
|||||||
* - UNCLOSED_STRING if there is an unclosed string in the document.
|
* - UNCLOSED_STRING if there is an unclosed string in the document.
|
||||||
*/
|
*/
|
||||||
simdjson_warn_unused simdjson_result<document> iterate(padded_string_view json) & noexcept;
|
simdjson_warn_unused simdjson_result<document> iterate(padded_string_view json) & noexcept;
|
||||||
|
#ifdef SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
|
||||||
|
simdjson_warn_unused simdjson_result<document> iterate_allow_incomplete_json(padded_string_view json) & noexcept;
|
||||||
|
#endif // SIMDJSON_EXPERIMENTAL_ALLOW_INCOMPLETE_JSON
|
||||||
/** @overload simdjson_result<document> iterate(padded_string_view json) & noexcept */
|
/** @overload simdjson_result<document> iterate(padded_string_view json) & noexcept */
|
||||||
simdjson_warn_unused simdjson_result<document> iterate(const char *json, size_t len, size_t capacity) & noexcept;
|
simdjson_warn_unused simdjson_result<document> iterate(const char *json, size_t len, size_t capacity) & noexcept;
|
||||||
/** @overload simdjson_result<document> iterate(padded_string_view json) & noexcept */
|
/** @overload simdjson_result<document> iterate(padded_string_view json) & noexcept */
|
||||||
@@ -242,9 +245,9 @@ public:
|
|||||||
simdjson_result<document_stream> iterate_many(const char *buf, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept = delete;
|
simdjson_result<document_stream> iterate_many(const char *buf, size_t batch_size = DEFAULT_BATCH_SIZE) noexcept = delete;
|
||||||
|
|
||||||
/** The capacity of this parser (the largest document it can process). */
|
/** The capacity of this parser (the largest document it can process). */
|
||||||
simdjson_inline size_t capacity() const noexcept;
|
simdjson_pure simdjson_inline size_t capacity() const noexcept;
|
||||||
/** The maximum capacity of this parser (the largest document it is allowed to process). */
|
/** The maximum capacity of this parser (the largest document it is allowed to process). */
|
||||||
simdjson_inline size_t max_capacity() const noexcept;
|
simdjson_pure simdjson_inline size_t max_capacity() const noexcept;
|
||||||
simdjson_inline void set_max_capacity(size_t max_capacity) noexcept;
|
simdjson_inline void set_max_capacity(size_t max_capacity) noexcept;
|
||||||
/**
|
/**
|
||||||
* The maximum depth of this parser (the most deeply nested objects and arrays it can process).
|
* The maximum depth of this parser (the most deeply nested objects and arrays it can process).
|
||||||
@@ -252,7 +255,7 @@ public:
|
|||||||
* The document's instance current_depth() method should be used to monitor the parsing
|
* The document's instance current_depth() method should be used to monitor the parsing
|
||||||
* depth and limit it if desired.
|
* depth and limit it if desired.
|
||||||
*/
|
*/
|
||||||
simdjson_inline size_t max_depth() const noexcept;
|
simdjson_pure simdjson_inline size_t max_depth() const noexcept;
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Ensure this parser has enough memory to process JSON documents up to `capacity` bytes in length
|
* Ensure this parser has enough memory to process JSON documents up to `capacity` bytes in length
|
||||||
@@ -275,8 +278,12 @@ public:
|
|||||||
* behavior of the parser for future operations.
|
* behavior of the parser for future operations.
|
||||||
*/
|
*/
|
||||||
bool threaded{true};
|
bool threaded{true};
|
||||||
|
#else
|
||||||
|
/**
|
||||||
|
* When SIMDJSON_THREADS_ENABLED is not defined, the parser instance cannot use threads.
|
||||||
|
*/
|
||||||
|
bool threaded{false};
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Unescape this JSON string, replacing \\ with \, \n with newline, etc. to a user-provided buffer.
|
* Unescape this JSON string, replacing \\ with \, \n with newline, etc. to a user-provided buffer.
|
||||||
* The result must be valid UTF-8.
|
* The result must be valid UTF-8.
|
||||||
@@ -324,9 +331,20 @@ public:
|
|||||||
*/
|
*/
|
||||||
simdjson_inline simdjson_result<std::string_view> unescape_wobbly(raw_json_string in, uint8_t *&dst) const noexcept;
|
simdjson_inline simdjson_result<std::string_view> unescape_wobbly(raw_json_string in, uint8_t *&dst) const noexcept;
|
||||||
|
|
||||||
|
#if SIMDJSON_DEVELOPMENT_CHECKS
|
||||||
|
/**
|
||||||
|
* Returns true if string_buf_loc is outside of the allocated range for the
|
||||||
|
* the string buffer. When true, it indicates that the string buffer has overflowed.
|
||||||
|
* This is a development-time check that is not needed in production. It can be
|
||||||
|
* used to detect buffer overflows in the string buffer and usafe usage of the
|
||||||
|
* string buffer.
|
||||||
|
*/
|
||||||
|
bool string_buffer_overflow(const uint8_t *string_buf_loc) const noexcept;
|
||||||
|
#endif
|
||||||
|
|
||||||
private:
|
private:
|
||||||
/** @private [for benchmarking access] The implementation to use */
|
/** @private [for benchmarking access] The implementation to use */
|
||||||
std::unique_ptr<internal::dom_parser_implementation> implementation{};
|
std::unique_ptr<simdjson::internal::dom_parser_implementation> implementation{};
|
||||||
size_t _capacity{0};
|
size_t _capacity{0};
|
||||||
size_t _max_capacity;
|
size_t _max_capacity;
|
||||||
size_t _max_depth{DEFAULT_MAX_DEPTH};
|
size_t _max_depth{DEFAULT_MAX_DEPTH};
|
||||||
|
|||||||
@@ -6,8 +6,6 @@
|
|||||||
#include "simdjson/generic/ondemand/array.h"
|
#include "simdjson/generic/ondemand/array.h"
|
||||||
#include "simdjson/generic/ondemand/array_iterator.h"
|
#include "simdjson/generic/ondemand/array_iterator.h"
|
||||||
#include "simdjson/generic/ondemand/json_iterator.h"
|
#include "simdjson/generic/ondemand/json_iterator.h"
|
||||||
#include "simdjson/generic/ondemand/json_path_to_pointer_conversion.h"
|
|
||||||
#include "simdjson/generic/ondemand/json_path_to_pointer_conversion-inl.h"
|
|
||||||
#include "simdjson/generic/ondemand/json_type.h"
|
#include "simdjson/generic/ondemand/json_type.h"
|
||||||
#include "simdjson/generic/ondemand/object.h"
|
#include "simdjson/generic/ondemand/object.h"
|
||||||
#include "simdjson/generic/ondemand/raw_json_string.h"
|
#include "simdjson/generic/ondemand/raw_json_string.h"
|
||||||
@@ -239,6 +237,26 @@ simdjson_inline int32_t value::current_depth() const noexcept{
|
|||||||
return iter.json_iter().depth();
|
return iter.json_iter().depth();
|
||||||
}
|
}
|
||||||
|
|
||||||
|
inline bool is_pointer_well_formed(std::string_view json_pointer) noexcept {
|
||||||
|
if (simdjson_unlikely(json_pointer.empty())) { // can't be
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
if (simdjson_unlikely(json_pointer[0] != '/')) {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
size_t escape = json_pointer.find('~');
|
||||||
|
if (escape == std::string_view::npos) {
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
if (escape == json_pointer.size() - 1) {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
if (json_pointer[escape + 1] != '0' && json_pointer[escape + 1] != '1') {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
simdjson_inline simdjson_result<value> value::at_pointer(std::string_view json_pointer) noexcept {
|
simdjson_inline simdjson_result<value> value::at_pointer(std::string_view json_pointer) noexcept {
|
||||||
json_type t;
|
json_type t;
|
||||||
SIMDJSON_TRY(type().get(t));
|
SIMDJSON_TRY(type().get(t));
|
||||||
@@ -249,6 +267,10 @@ simdjson_inline simdjson_result<value> value::at_pointer(std::string_view json_p
|
|||||||
case json_type::object:
|
case json_type::object:
|
||||||
return (*this).get_object().at_pointer(json_pointer);
|
return (*this).get_object().at_pointer(json_pointer);
|
||||||
default:
|
default:
|
||||||
|
// a non-empty string can be invalid, or accessing a primitive (issue 2154)
|
||||||
|
if (is_pointer_well_formed(json_pointer)) {
|
||||||
|
return NO_SUCH_FIELD;
|
||||||
|
}
|
||||||
return INVALID_JSON_POINTER;
|
return INVALID_JSON_POINTER;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -6,9 +6,9 @@
|
|||||||
#include "simdjson/generic/atomparsing.h"
|
#include "simdjson/generic/atomparsing.h"
|
||||||
#include "simdjson/generic/numberparsing.h"
|
#include "simdjson/generic/numberparsing.h"
|
||||||
#include "simdjson/generic/ondemand/json_iterator.h"
|
#include "simdjson/generic/ondemand/json_iterator.h"
|
||||||
|
#include "simdjson/generic/ondemand/value_iterator.h"
|
||||||
#include "simdjson/generic/ondemand/json_type-inl.h"
|
#include "simdjson/generic/ondemand/json_type-inl.h"
|
||||||
#include "simdjson/generic/ondemand/raw_json_string-inl.h"
|
#include "simdjson/generic/ondemand/raw_json_string-inl.h"
|
||||||
#include "simdjson/generic/ondemand/value_iterator.h"
|
|
||||||
#endif // SIMDJSON_CONDITIONAL_INCLUDE
|
#endif // SIMDJSON_CONDITIONAL_INCLUDE
|
||||||
|
|
||||||
namespace simdjson {
|
namespace simdjson {
|
||||||
@@ -799,6 +799,8 @@ simdjson_inline simdjson_result<bool> value_iterator::is_root_null(bool check_tr
|
|||||||
if(result) { // we have something that looks like a null.
|
if(result) { // we have something that looks like a null.
|
||||||
if (check_trailing && !_json_iter->is_single_token()) { return TRAILING_CONTENT; }
|
if (check_trailing && !_json_iter->is_single_token()) { return TRAILING_CONTENT; }
|
||||||
advance_root_scalar("null");
|
advance_root_scalar("null");
|
||||||
|
} else if (json[0] == 'n') {
|
||||||
|
return incorrect_type_error("Not a null but starts with n");
|
||||||
}
|
}
|
||||||
return result;
|
return result;
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -4,8 +4,14 @@
|
|||||||
#include "simdjson/haswell/intrinsics.h"
|
#include "simdjson/haswell/intrinsics.h"
|
||||||
|
|
||||||
#if !SIMDJSON_CAN_ALWAYS_RUN_HASWELL
|
#if !SIMDJSON_CAN_ALWAYS_RUN_HASWELL
|
||||||
|
// We enable bmi2 only if LLVM/clang is used, because GCC may not
|
||||||
|
// make good use of it. See https://github.com/simdjson/simdjson/pull/2243
|
||||||
|
#if defined(__clang__)
|
||||||
|
SIMDJSON_TARGET_REGION("avx2,bmi,bmi2,pclmul,lzcnt,popcnt")
|
||||||
|
#else
|
||||||
SIMDJSON_TARGET_REGION("avx2,bmi,pclmul,lzcnt,popcnt")
|
SIMDJSON_TARGET_REGION("avx2,bmi,pclmul,lzcnt,popcnt")
|
||||||
#endif
|
#endif
|
||||||
|
#endif
|
||||||
|
|
||||||
#include "simdjson/haswell/bitmanipulation.h"
|
#include "simdjson/haswell/bitmanipulation.h"
|
||||||
#include "simdjson/haswell/bitmask.h"
|
#include "simdjson/haswell/bitmask.h"
|
||||||
|
|||||||
@@ -37,13 +37,13 @@ static simdjson_inline uint32_t parse_eight_digits_unrolled(const uint8_t *chars
|
|||||||
simdjson_inline internal::value128 full_multiplication(uint64_t value1, uint64_t value2) {
|
simdjson_inline internal::value128 full_multiplication(uint64_t value1, uint64_t value2) {
|
||||||
internal::value128 answer;
|
internal::value128 answer;
|
||||||
#if SIMDJSON_REGULAR_VISUAL_STUDIO || SIMDJSON_IS_32BITS
|
#if SIMDJSON_REGULAR_VISUAL_STUDIO || SIMDJSON_IS_32BITS
|
||||||
#ifdef _M_ARM64
|
#if SIMDJSON_IS_ARM64
|
||||||
// ARM64 has native support for 64-bit multiplications, no need to emultate
|
// ARM64 has native support for 64-bit multiplications, no need to emultate
|
||||||
answer.high = __umulh(value1, value2);
|
answer.high = __umulh(value1, value2);
|
||||||
answer.low = value1 * value2;
|
answer.low = value1 * value2;
|
||||||
#else
|
#else
|
||||||
answer.low = _umul128(value1, value2, &answer.high); // _umul128 not available on ARM64
|
answer.low = _umul128(value1, value2, &answer.high); // _umul128 not available on ARM64
|
||||||
#endif // _M_ARM64
|
#endif // SIMDJSON_IS_ARM64
|
||||||
#else // SIMDJSON_REGULAR_VISUAL_STUDIO || SIMDJSON_IS_32BITS
|
#else // SIMDJSON_REGULAR_VISUAL_STUDIO || SIMDJSON_IS_32BITS
|
||||||
__uint128_t r = (static_cast<__uint128_t>(value1)) * value2;
|
__uint128_t r = (static_cast<__uint128_t>(value1)) * value2;
|
||||||
answer.low = uint64_t(r);
|
answer.low = uint64_t(r);
|
||||||
|
|||||||
@@ -65,10 +65,10 @@ namespace simd {
|
|||||||
struct simd8<bool>: base8<bool> {
|
struct simd8<bool>: base8<bool> {
|
||||||
static simdjson_inline simd8<bool> splat(bool _value) { return _mm256_set1_epi8(uint8_t(-(!!_value))); }
|
static simdjson_inline simd8<bool> splat(bool _value) { return _mm256_set1_epi8(uint8_t(-(!!_value))); }
|
||||||
|
|
||||||
simdjson_inline simd8<bool>() : base8() {}
|
simdjson_inline simd8() : base8() {}
|
||||||
simdjson_inline simd8<bool>(const __m256i _value) : base8<bool>(_value) {}
|
simdjson_inline simd8(const __m256i _value) : base8<bool>(_value) {}
|
||||||
// Splat constructor
|
// Splat constructor
|
||||||
simdjson_inline simd8<bool>(bool _value) : base8<bool>(splat(_value)) {}
|
simdjson_inline simd8(bool _value) : base8<bool>(splat(_value)) {}
|
||||||
|
|
||||||
simdjson_inline int to_bitmask() const { return _mm256_movemask_epi8(*this); }
|
simdjson_inline int to_bitmask() const { return _mm256_movemask_epi8(*this); }
|
||||||
simdjson_inline bool any() const { return !_mm256_testz_si256(*this, *this); }
|
simdjson_inline bool any() const { return !_mm256_testz_si256(*this, *this); }
|
||||||
|
|||||||
@@ -33,13 +33,13 @@ static simdjson_inline uint32_t parse_eight_digits_unrolled(const uint8_t *chars
|
|||||||
simdjson_inline internal::value128 full_multiplication(uint64_t value1, uint64_t value2) {
|
simdjson_inline internal::value128 full_multiplication(uint64_t value1, uint64_t value2) {
|
||||||
internal::value128 answer;
|
internal::value128 answer;
|
||||||
#if SIMDJSON_REGULAR_VISUAL_STUDIO || SIMDJSON_IS_32BITS
|
#if SIMDJSON_REGULAR_VISUAL_STUDIO || SIMDJSON_IS_32BITS
|
||||||
#ifdef _M_ARM64
|
#if SIMDJSON_IS_ARM64
|
||||||
// ARM64 has native support for 64-bit multiplications, no need to emultate
|
// ARM64 has native support for 64-bit multiplications, no need to emultate
|
||||||
answer.high = __umulh(value1, value2);
|
answer.high = __umulh(value1, value2);
|
||||||
answer.low = value1 * value2;
|
answer.low = value1 * value2;
|
||||||
#else
|
#else
|
||||||
answer.low = _umul128(value1, value2, &answer.high); // _umul128 not available on ARM64
|
answer.low = _umul128(value1, value2, &answer.high); // _umul128 not available on ARM64
|
||||||
#endif // _M_ARM64
|
#endif // SIMDJSON_IS_ARM64
|
||||||
#else // SIMDJSON_REGULAR_VISUAL_STUDIO || SIMDJSON_IS_32BITS
|
#else // SIMDJSON_REGULAR_VISUAL_STUDIO || SIMDJSON_IS_32BITS
|
||||||
__uint128_t r = (static_cast<__uint128_t>(value1)) * value2;
|
__uint128_t r = (static_cast<__uint128_t>(value1)) * value2;
|
||||||
answer.low = uint64_t(r);
|
answer.low = uint64_t(r);
|
||||||
|
|||||||
@@ -93,10 +93,10 @@ namespace simd {
|
|||||||
struct simd8<bool>: base8<bool> {
|
struct simd8<bool>: base8<bool> {
|
||||||
static simdjson_inline simd8<bool> splat(bool _value) { return _mm512_set1_epi8(uint8_t(-(!!_value))); }
|
static simdjson_inline simd8<bool> splat(bool _value) { return _mm512_set1_epi8(uint8_t(-(!!_value))); }
|
||||||
|
|
||||||
simdjson_inline simd8<bool>() : base8() {}
|
simdjson_inline simd8() : base8() {}
|
||||||
simdjson_inline simd8<bool>(const __m512i _value) : base8<bool>(_value) {}
|
simdjson_inline simd8(const __m512i _value) : base8<bool>(_value) {}
|
||||||
// Splat constructor
|
// Splat constructor
|
||||||
simdjson_inline simd8<bool>(bool _value) : base8<bool>(splat(_value)) {}
|
simdjson_inline simd8(bool _value) : base8<bool>(splat(_value)) {}
|
||||||
simdjson_inline bool any() const { return !!_mm512_test_epi8_mask (*this, *this); }
|
simdjson_inline bool any() const { return !!_mm512_test_epi8_mask (*this, *this); }
|
||||||
simdjson_inline simd8<bool> operator~() const { return *this ^ true; }
|
simdjson_inline simd8<bool> operator~() const { return *this ^ true; }
|
||||||
};
|
};
|
||||||
|
|||||||
@@ -53,7 +53,7 @@ public:
|
|||||||
*
|
*
|
||||||
* @return the name of the implementation, e.g. "haswell", "westmere", "arm64".
|
* @return the name of the implementation, e.g. "haswell", "westmere", "arm64".
|
||||||
*/
|
*/
|
||||||
virtual const std::string &name() const { return _name; }
|
virtual std::string name() const { return std::string(_name); }
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* The description of this implementation.
|
* The description of this implementation.
|
||||||
@@ -63,7 +63,7 @@ public:
|
|||||||
*
|
*
|
||||||
* @return the description of the implementation, e.g. "Intel/AMD AVX2", "Intel/AMD SSE4.2", "ARM NEON".
|
* @return the description of the implementation, e.g. "Intel/AMD AVX2", "Intel/AMD SSE4.2", "ARM NEON".
|
||||||
*/
|
*/
|
||||||
virtual const std::string &description() const { return _description; }
|
virtual std::string description() const { return std::string(_description); }
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* The instruction sets this implementation is compiled against
|
* The instruction sets this implementation is compiled against
|
||||||
@@ -139,18 +139,19 @@ protected:
|
|||||||
_required_instruction_sets(required_instruction_sets)
|
_required_instruction_sets(required_instruction_sets)
|
||||||
{
|
{
|
||||||
}
|
}
|
||||||
virtual ~implementation()=default;
|
protected:
|
||||||
|
~implementation() = default;
|
||||||
|
|
||||||
private:
|
private:
|
||||||
/**
|
/**
|
||||||
* The name of this implementation.
|
* The name of this implementation.
|
||||||
*/
|
*/
|
||||||
const std::string _name;
|
std::string_view _name;
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* The description of this implementation.
|
* The description of this implementation.
|
||||||
*/
|
*/
|
||||||
const std::string _description;
|
std::string_view _description;
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Instruction sets required for this implementation.
|
* Instruction sets required for this implementation.
|
||||||
|
|||||||
@@ -106,7 +106,7 @@ public:
|
|||||||
* Unescape a valid UTF-8 string from src to dst, stopping at a final unescaped quote. There
|
* Unescape a valid UTF-8 string from src to dst, stopping at a final unescaped quote. There
|
||||||
* must be an unescaped quote terminating the string. It returns the final output
|
* must be an unescaped quote terminating the string. It returns the final output
|
||||||
* position as pointer. In case of error (e.g., the string has bad escaped codes),
|
* position as pointer. In case of error (e.g., the string has bad escaped codes),
|
||||||
* then null_nullptrptr is returned. It is assumed that the output buffer is large
|
* then null_ptr is returned. It is assumed that the output buffer is large
|
||||||
* enough. E.g., if src points at 'joe"', then dst needs to have four free bytes +
|
* enough. E.g., if src points at 'joe"', then dst needs to have four free bytes +
|
||||||
* SIMDJSON_PADDING bytes.
|
* SIMDJSON_PADDING bytes.
|
||||||
*
|
*
|
||||||
@@ -123,7 +123,7 @@ public:
|
|||||||
* Unescape a NON-valid UTF-8 string from src to dst, stopping at a final unescaped quote. There
|
* Unescape a NON-valid UTF-8 string from src to dst, stopping at a final unescaped quote. There
|
||||||
* must be an unescaped quote terminating the string. It returns the final output
|
* must be an unescaped quote terminating the string. It returns the final output
|
||||||
* position as pointer. In case of error (e.g., the string has bad escaped codes),
|
* position as pointer. In case of error (e.g., the string has bad escaped codes),
|
||||||
* then null_nullptrptr is returned. It is assumed that the output buffer is large
|
* then null_ptr is returned. It is assumed that the output buffer is large
|
||||||
* enough. E.g., if src points at 'joe"', then dst needs to have four free bytes +
|
* enough. E.g., if src points at 'joe"', then dst needs to have four free bytes +
|
||||||
* SIMDJSON_PADDING bytes.
|
* SIMDJSON_PADDING bytes.
|
||||||
*
|
*
|
||||||
@@ -177,14 +177,14 @@ public:
|
|||||||
*
|
*
|
||||||
* @return Current capacity, in bytes.
|
* @return Current capacity, in bytes.
|
||||||
*/
|
*/
|
||||||
simdjson_inline size_t capacity() const noexcept;
|
simdjson_pure simdjson_inline size_t capacity() const noexcept;
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* The maximum level of nested object and arrays supported by this parser.
|
* The maximum level of nested object and arrays supported by this parser.
|
||||||
*
|
*
|
||||||
* @return Maximum depth, in bytes.
|
* @return Maximum depth, in bytes.
|
||||||
*/
|
*/
|
||||||
simdjson_inline size_t max_depth() const noexcept;
|
simdjson_pure simdjson_inline size_t max_depth() const noexcept;
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Ensure this parser has enough memory to process JSON documents up to `capacity` bytes in length
|
* Ensure this parser has enough memory to process JSON documents up to `capacity` bytes in length
|
||||||
@@ -225,11 +225,11 @@ simdjson_inline dom_parser_implementation::dom_parser_implementation() noexcept
|
|||||||
simdjson_inline dom_parser_implementation::dom_parser_implementation(dom_parser_implementation &&other) noexcept = default;
|
simdjson_inline dom_parser_implementation::dom_parser_implementation(dom_parser_implementation &&other) noexcept = default;
|
||||||
simdjson_inline dom_parser_implementation &dom_parser_implementation::operator=(dom_parser_implementation &&other) noexcept = default;
|
simdjson_inline dom_parser_implementation &dom_parser_implementation::operator=(dom_parser_implementation &&other) noexcept = default;
|
||||||
|
|
||||||
simdjson_inline size_t dom_parser_implementation::capacity() const noexcept {
|
simdjson_pure simdjson_inline size_t dom_parser_implementation::capacity() const noexcept {
|
||||||
return _capacity;
|
return _capacity;
|
||||||
}
|
}
|
||||||
|
|
||||||
simdjson_inline size_t dom_parser_implementation::max_depth() const noexcept {
|
simdjson_pure simdjson_inline size_t dom_parser_implementation::max_depth() const noexcept {
|
||||||
return _max_depth;
|
return _max_depth;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,64 @@
|
|||||||
|
#ifndef SIMDJSON_JSONPATHUTIL_H
|
||||||
|
#define SIMDJSON_JSONPATHUTIL_H
|
||||||
|
|
||||||
|
#include <string>
|
||||||
|
#include <string_view>
|
||||||
|
|
||||||
|
namespace simdjson {
|
||||||
|
/**
|
||||||
|
* Converts JSONPath to JSON Pointer.
|
||||||
|
* @param json_path The JSONPath string to be converted.
|
||||||
|
* @return A string containing the equivalent JSON Pointer.
|
||||||
|
*/
|
||||||
|
inline std::string json_path_to_pointer_conversion(std::string_view json_path) {
|
||||||
|
size_t i = 0;
|
||||||
|
|
||||||
|
// if JSONPath starts with $, skip it
|
||||||
|
if (!json_path.empty() && json_path.front() == '$') {
|
||||||
|
i = 1;
|
||||||
|
}
|
||||||
|
if (json_path.empty() || (json_path[i] != '.' &&
|
||||||
|
json_path[i] != '[')) {
|
||||||
|
return "-1"; // This is just a sentinel value, the caller should check for this and return an error.
|
||||||
|
}
|
||||||
|
|
||||||
|
std::string result;
|
||||||
|
// Reserve space to reduce allocations, adjusting for potential increases due
|
||||||
|
// to escaping.
|
||||||
|
result.reserve(json_path.size() * 2);
|
||||||
|
|
||||||
|
while (i < json_path.length()) {
|
||||||
|
if (json_path[i] == '.') {
|
||||||
|
result += '/';
|
||||||
|
} else if (json_path[i] == '[') {
|
||||||
|
result += '/';
|
||||||
|
++i; // Move past the '['
|
||||||
|
while (i < json_path.length() && json_path[i] != ']') {
|
||||||
|
if (json_path[i] == '~') {
|
||||||
|
result += "~0";
|
||||||
|
} else if (json_path[i] == '/') {
|
||||||
|
result += "~1";
|
||||||
|
} else {
|
||||||
|
result += json_path[i];
|
||||||
|
}
|
||||||
|
++i;
|
||||||
|
}
|
||||||
|
if (i == json_path.length() || json_path[i] != ']') {
|
||||||
|
return "-1"; // Using sentinel value that will be handled as an error by the caller.
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
if (json_path[i] == '~') {
|
||||||
|
result += "~0";
|
||||||
|
} else if (json_path[i] == '/') {
|
||||||
|
result += "~1";
|
||||||
|
} else {
|
||||||
|
result += json_path[i];
|
||||||
|
}
|
||||||
|
}
|
||||||
|
++i;
|
||||||
|
}
|
||||||
|
|
||||||
|
return result;
|
||||||
|
}
|
||||||
|
} // namespace simdjson
|
||||||
|
#endif // SIMDJSON_JSONPATHUTIL_H
|
||||||
@@ -36,6 +36,12 @@ simdjson_inline internal::value128 full_multiplication(uint64_t value1, uint64_t
|
|||||||
} // namespace lasx
|
} // namespace lasx
|
||||||
} // namespace simdjson
|
} // namespace simdjson
|
||||||
|
|
||||||
|
#ifndef SIMDJSON_SWAR_NUMBER_PARSING
|
||||||
|
#if SIMDJSON_IS_BIG_ENDIAN
|
||||||
|
#define SIMDJSON_SWAR_NUMBER_PARSING 0
|
||||||
|
#else
|
||||||
#define SIMDJSON_SWAR_NUMBER_PARSING 1
|
#define SIMDJSON_SWAR_NUMBER_PARSING 1
|
||||||
|
#endif
|
||||||
|
#endif
|
||||||
|
|
||||||
#endif // SIMDJSON_LASX_NUMBERPARSING_DEFS_H
|
#endif // SIMDJSON_LASX_NUMBERPARSING_DEFS_H
|
||||||
|
|||||||
@@ -67,10 +67,10 @@ namespace simd {
|
|||||||
struct simd8<bool>: base8<bool> {
|
struct simd8<bool>: base8<bool> {
|
||||||
static simdjson_inline simd8<bool> splat(bool _value) { return __lasx_xvreplgr2vr_b(uint8_t(-(!!_value))); }
|
static simdjson_inline simd8<bool> splat(bool _value) { return __lasx_xvreplgr2vr_b(uint8_t(-(!!_value))); }
|
||||||
|
|
||||||
simdjson_inline simd8<bool>() : base8() {}
|
simdjson_inline simd8() : base8() {}
|
||||||
simdjson_inline simd8<bool>(const __m256i _value) : base8<bool>(_value) {}
|
simdjson_inline simd8(const __m256i _value) : base8<bool>(_value) {}
|
||||||
// Splat constructor
|
// Splat constructor
|
||||||
simdjson_inline simd8<bool>(bool _value) : base8<bool>(splat(_value)) {}
|
simdjson_inline simd8(bool _value) : base8<bool>(splat(_value)) {}
|
||||||
|
|
||||||
simdjson_inline int to_bitmask() const {
|
simdjson_inline int to_bitmask() const {
|
||||||
__m256i mask = __lasx_xvmskltz_b(*this);
|
__m256i mask = __lasx_xvmskltz_b(*this);
|
||||||
|
|||||||
@@ -36,6 +36,12 @@ simdjson_inline internal::value128 full_multiplication(uint64_t value1, uint64_t
|
|||||||
} // namespace lsx
|
} // namespace lsx
|
||||||
} // namespace simdjson
|
} // namespace simdjson
|
||||||
|
|
||||||
|
#ifndef SIMDJSON_SWAR_NUMBER_PARSING
|
||||||
|
#if SIMDJSON_IS_BIG_ENDIAN
|
||||||
|
#define SIMDJSON_SWAR_NUMBER_PARSING 0
|
||||||
|
#else
|
||||||
#define SIMDJSON_SWAR_NUMBER_PARSING 1
|
#define SIMDJSON_SWAR_NUMBER_PARSING 1
|
||||||
|
#endif
|
||||||
|
#endif
|
||||||
|
|
||||||
#endif // SIMDJSON_LSX_NUMBERPARSING_DEFS_H
|
#endif // SIMDJSON_LSX_NUMBERPARSING_DEFS_H
|
||||||
|
|||||||
@@ -65,10 +65,10 @@ namespace simd {
|
|||||||
return __lsx_vreplgr2vr_b(uint8_t(-(!!_value)));
|
return __lsx_vreplgr2vr_b(uint8_t(-(!!_value)));
|
||||||
}
|
}
|
||||||
|
|
||||||
simdjson_inline simd8<bool>() : base8() {}
|
simdjson_inline simd8() : base8() {}
|
||||||
simdjson_inline simd8<bool>(const __m128i _value) : base8<bool>(_value) {}
|
simdjson_inline simd8(const __m128i _value) : base8<bool>(_value) {}
|
||||||
// Splat constructor
|
// Splat constructor
|
||||||
simdjson_inline simd8<bool>(bool _value) : base8<bool>(splat(_value)) {}
|
simdjson_inline simd8(bool _value) : base8<bool>(splat(_value)) {}
|
||||||
|
|
||||||
simdjson_inline int to_bitmask() const { return __lsx_vpickve2gr_w(__lsx_vmskltz_b(*this), 0); }
|
simdjson_inline int to_bitmask() const { return __lsx_vpickve2gr_w(__lsx_vmskltz_b(*this), 0); }
|
||||||
simdjson_inline bool any() const { return 0 == __lsx_vpickve2gr_hu(__lsx_vmsknz_b(*this), 0); }
|
simdjson_inline bool any() const { return 0 == __lsx_vpickve2gr_hu(__lsx_vmsknz_b(*this), 0); }
|
||||||
|
|||||||
@@ -6,13 +6,13 @@
|
|||||||
// Distributed under the Boost Software License, Version 1.0.
|
// Distributed under the Boost Software License, Version 1.0.
|
||||||
// (See accompanying file LICENSE.txt or copy at http://www.boost.org/LICENSE_1_0.txt)
|
// (See accompanying file LICENSE.txt or copy at http://www.boost.org/LICENSE_1_0.txt)
|
||||||
|
|
||||||
// #pragma once // We remove #pragma once here as it generates a warning in some cases. We rely on the include guard.
|
#pragma once
|
||||||
|
|
||||||
#ifndef NONSTD_SV_LITE_H_INCLUDED
|
#ifndef NONSTD_SV_LITE_H_INCLUDED
|
||||||
#define NONSTD_SV_LITE_H_INCLUDED
|
#define NONSTD_SV_LITE_H_INCLUDED
|
||||||
|
|
||||||
#define string_view_lite_MAJOR 1
|
#define string_view_lite_MAJOR 1
|
||||||
#define string_view_lite_MINOR 7
|
#define string_view_lite_MINOR 8
|
||||||
#define string_view_lite_PATCH 0
|
#define string_view_lite_PATCH 0
|
||||||
|
|
||||||
#define string_view_lite_VERSION nssv_STRINGIFY(string_view_lite_MAJOR) "." nssv_STRINGIFY(string_view_lite_MINOR) "." nssv_STRINGIFY(string_view_lite_PATCH)
|
#define string_view_lite_VERSION nssv_STRINGIFY(string_view_lite_MAJOR) "." nssv_STRINGIFY(string_view_lite_MINOR) "." nssv_STRINGIFY(string_view_lite_PATCH)
|
||||||
@@ -134,6 +134,8 @@
|
|||||||
|
|
||||||
#if nssv_CONFIG_CONVERSION_STD_STRING_FREE_FUNCTIONS
|
#if nssv_CONFIG_CONVERSION_STD_STRING_FREE_FUNCTIONS
|
||||||
|
|
||||||
|
#include <string>
|
||||||
|
|
||||||
namespace nonstd {
|
namespace nonstd {
|
||||||
|
|
||||||
template< class CharT, class Traits, class Allocator = std::allocator<CharT> >
|
template< class CharT, class Traits, class Allocator = std::allocator<CharT> >
|
||||||
|
|||||||
@@ -53,6 +53,9 @@ inline padded_string::padded_string(const char *data, size_t length) noexcept
|
|||||||
if ((data != nullptr) && (data_ptr != nullptr)) {
|
if ((data != nullptr) && (data_ptr != nullptr)) {
|
||||||
std::memcpy(data_ptr, data, length);
|
std::memcpy(data_ptr, data, length);
|
||||||
}
|
}
|
||||||
|
if (data_ptr == nullptr) {
|
||||||
|
viable_size = 0;
|
||||||
|
}
|
||||||
}
|
}
|
||||||
#ifdef __cpp_char8_t
|
#ifdef __cpp_char8_t
|
||||||
inline padded_string::padded_string(const char8_t *data, size_t length) noexcept
|
inline padded_string::padded_string(const char8_t *data, size_t length) noexcept
|
||||||
@@ -60,12 +63,17 @@ inline padded_string::padded_string(const char8_t *data, size_t length) noexcept
|
|||||||
if ((data != nullptr) && (data_ptr != nullptr)) {
|
if ((data != nullptr) && (data_ptr != nullptr)) {
|
||||||
std::memcpy(data_ptr, reinterpret_cast<const char *>(data), length);
|
std::memcpy(data_ptr, reinterpret_cast<const char *>(data), length);
|
||||||
}
|
}
|
||||||
|
if (data_ptr == nullptr) {
|
||||||
|
viable_size = 0;
|
||||||
|
}
|
||||||
}
|
}
|
||||||
#endif
|
#endif
|
||||||
// note: do not pass std::string arguments by value
|
// note: do not pass std::string arguments by value
|
||||||
inline padded_string::padded_string(const std::string & str_ ) noexcept
|
inline padded_string::padded_string(const std::string & str_ ) noexcept
|
||||||
: viable_size(str_.size()), data_ptr(internal::allocate_padded_buffer(str_.size())) {
|
: viable_size(str_.size()), data_ptr(internal::allocate_padded_buffer(str_.size())) {
|
||||||
if (data_ptr != nullptr) {
|
if (data_ptr == nullptr) {
|
||||||
|
viable_size = 0;
|
||||||
|
} else {
|
||||||
std::memcpy(data_ptr, str_.data(), str_.size());
|
std::memcpy(data_ptr, str_.data(), str_.size());
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -84,6 +92,7 @@ inline padded_string::padded_string(std::string_view sv_) noexcept
|
|||||||
inline padded_string::padded_string(padded_string &&o) noexcept
|
inline padded_string::padded_string(padded_string &&o) noexcept
|
||||||
: viable_size(o.viable_size), data_ptr(o.data_ptr) {
|
: viable_size(o.viable_size), data_ptr(o.data_ptr) {
|
||||||
o.data_ptr = nullptr; // we take ownership
|
o.data_ptr = nullptr; // we take ownership
|
||||||
|
o.viable_size = 0;
|
||||||
}
|
}
|
||||||
|
|
||||||
inline padded_string &padded_string::operator=(padded_string &&o) noexcept {
|
inline padded_string &padded_string::operator=(padded_string &&o) noexcept {
|
||||||
|
|||||||
@@ -11,6 +11,9 @@
|
|||||||
#include <strings.h>
|
#include <strings.h>
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
|
// We are using size_t without namespace std:: throughout the project
|
||||||
|
using std::size_t;
|
||||||
|
|
||||||
#ifdef _MSC_VER
|
#ifdef _MSC_VER
|
||||||
#define SIMDJSON_VISUAL_STUDIO 1
|
#define SIMDJSON_VISUAL_STUDIO 1
|
||||||
/**
|
/**
|
||||||
@@ -32,9 +35,9 @@
|
|||||||
#endif // __clang__
|
#endif // __clang__
|
||||||
#endif // _MSC_VER
|
#endif // _MSC_VER
|
||||||
|
|
||||||
#if defined(__x86_64__) || defined(_M_AMD64)
|
#if (defined(__x86_64__) || defined(_M_AMD64)) && !defined(_M_ARM64EC)
|
||||||
#define SIMDJSON_IS_X86_64 1
|
#define SIMDJSON_IS_X86_64 1
|
||||||
#elif defined(__aarch64__) || defined(_M_ARM64)
|
#elif defined(__aarch64__) || defined(_M_ARM64) || defined(_M_ARM64EC)
|
||||||
#define SIMDJSON_IS_ARM64 1
|
#define SIMDJSON_IS_ARM64 1
|
||||||
#elif defined(__riscv) && __riscv_xlen == 64
|
#elif defined(__riscv) && __riscv_xlen == 64
|
||||||
#define SIMDJSON_IS_RISCV64 1
|
#define SIMDJSON_IS_RISCV64 1
|
||||||
@@ -148,6 +151,11 @@
|
|||||||
#define SIMDJSON_NO_SANITIZE_UNDEFINED
|
#define SIMDJSON_NO_SANITIZE_UNDEFINED
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
|
#if defined(__clang__) || defined(__GNUC__)
|
||||||
|
#define simdjson_pure [[gnu::pure]]
|
||||||
|
#else
|
||||||
|
#define simdjson_pure
|
||||||
|
#endif
|
||||||
|
|
||||||
#if defined(__clang__) || defined(__GNUC__)
|
#if defined(__clang__) || defined(__GNUC__)
|
||||||
#if defined(__has_feature)
|
#if defined(__has_feature)
|
||||||
@@ -193,4 +201,43 @@
|
|||||||
|
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
#if defined __BYTE_ORDER__ && defined __ORDER_BIG_ENDIAN__
|
||||||
|
#define SIMDJSON_IS_BIG_ENDIAN (__BYTE_ORDER__ == __ORDER_BIG_ENDIAN__)
|
||||||
|
#elif defined _WIN32
|
||||||
|
#define SIMDJSON_IS_BIG_ENDIAN 0
|
||||||
|
#else
|
||||||
|
#if defined(__APPLE__) || defined(__FreeBSD__)
|
||||||
|
#include <machine/endian.h>
|
||||||
|
#elif defined(sun) || defined(__sun)
|
||||||
|
#include <sys/byteorder.h>
|
||||||
|
#elif defined(__MVS__)
|
||||||
|
#include <sys/endian.h>
|
||||||
|
#else
|
||||||
|
#ifdef __has_include
|
||||||
|
#if __has_include(<endian.h>)
|
||||||
|
#include <endian.h>
|
||||||
|
#endif //__has_include(<endian.h>)
|
||||||
|
#endif //__has_include
|
||||||
|
#endif
|
||||||
|
#
|
||||||
|
#ifndef __BYTE_ORDER__
|
||||||
|
// safe choice
|
||||||
|
#define SIMDJSON_IS_BIG_ENDIAN 0
|
||||||
|
#endif
|
||||||
|
#
|
||||||
|
#ifndef __ORDER_LITTLE_ENDIAN__
|
||||||
|
// safe choice
|
||||||
|
#define SIMDJSON_IS_BIG_ENDIAN 0
|
||||||
|
#endif
|
||||||
|
#
|
||||||
|
#if __BYTE_ORDER__ == __ORDER_LITTLE_ENDIAN__
|
||||||
|
#define SIMDJSON_IS_BIG_ENDIAN 0
|
||||||
|
#else
|
||||||
|
#define SIMDJSON_IS_BIG_ENDIAN 1
|
||||||
|
#endif
|
||||||
|
#endif
|
||||||
|
|
||||||
|
|
||||||
#endif // SIMDJSON_PORTABILITY_H
|
#endif // SIMDJSON_PORTABILITY_H
|
||||||
|
|||||||
@@ -41,13 +41,13 @@ static simdjson_inline uint32_t parse_eight_digits_unrolled(const uint8_t *chars
|
|||||||
simdjson_inline internal::value128 full_multiplication(uint64_t value1, uint64_t value2) {
|
simdjson_inline internal::value128 full_multiplication(uint64_t value1, uint64_t value2) {
|
||||||
internal::value128 answer;
|
internal::value128 answer;
|
||||||
#if SIMDJSON_REGULAR_VISUAL_STUDIO || SIMDJSON_IS_32BITS
|
#if SIMDJSON_REGULAR_VISUAL_STUDIO || SIMDJSON_IS_32BITS
|
||||||
#ifdef _M_ARM64
|
#if SIMDJSON_IS_ARM64
|
||||||
// ARM64 has native support for 64-bit multiplications, no need to emultate
|
// ARM64 has native support for 64-bit multiplications, no need to emultate
|
||||||
answer.high = __umulh(value1, value2);
|
answer.high = __umulh(value1, value2);
|
||||||
answer.low = value1 * value2;
|
answer.low = value1 * value2;
|
||||||
#else
|
#else
|
||||||
answer.low = _umul128(value1, value2, &answer.high); // _umul128 not available on ARM64
|
answer.low = _umul128(value1, value2, &answer.high); // _umul128 not available on ARM64
|
||||||
#endif // _M_ARM64
|
#endif // SIMDJSON_IS_ARM64
|
||||||
#else // SIMDJSON_REGULAR_VISUAL_STUDIO || SIMDJSON_IS_32BITS
|
#else // SIMDJSON_REGULAR_VISUAL_STUDIO || SIMDJSON_IS_32BITS
|
||||||
__uint128_t r = (static_cast<__uint128_t>(value1)) * value2;
|
__uint128_t r = (static_cast<__uint128_t>(value1)) * value2;
|
||||||
answer.low = uint64_t(r);
|
answer.low = uint64_t(r);
|
||||||
@@ -60,6 +60,12 @@ simdjson_inline internal::value128 full_multiplication(uint64_t value1, uint64_t
|
|||||||
} // namespace ppc64
|
} // namespace ppc64
|
||||||
} // namespace simdjson
|
} // namespace simdjson
|
||||||
|
|
||||||
|
#ifndef SIMDJSON_SWAR_NUMBER_PARSING
|
||||||
|
#if SIMDJSON_IS_BIG_ENDIAN
|
||||||
|
#define SIMDJSON_SWAR_NUMBER_PARSING 0
|
||||||
|
#else
|
||||||
#define SIMDJSON_SWAR_NUMBER_PARSING 1
|
#define SIMDJSON_SWAR_NUMBER_PARSING 1
|
||||||
|
#endif
|
||||||
|
#endif
|
||||||
|
|
||||||
#endif // SIMDJSON_PPC64_NUMBERPARSING_DEFS_H
|
#endif // SIMDJSON_PPC64_NUMBERPARSING_DEFS_H
|
||||||
|
|||||||
@@ -96,11 +96,11 @@ template <> struct simd8<bool> : base8<bool> {
|
|||||||
return (__m128i)vec_splats((unsigned char)(-(!!_value)));
|
return (__m128i)vec_splats((unsigned char)(-(!!_value)));
|
||||||
}
|
}
|
||||||
|
|
||||||
simdjson_inline simd8<bool>() : base8<bool>() {}
|
simdjson_inline simd8() : base8<bool>() {}
|
||||||
simdjson_inline simd8<bool>(const __m128i _value)
|
simdjson_inline simd8(const __m128i _value)
|
||||||
: base8<bool>(_value) {}
|
: base8<bool>(_value) {}
|
||||||
// Splat constructor
|
// Splat constructor
|
||||||
simdjson_inline simd8<bool>(bool _value)
|
simdjson_inline simd8(bool _value)
|
||||||
: base8<bool>(splat(_value)) {}
|
: base8<bool>(splat(_value)) {}
|
||||||
|
|
||||||
simdjson_inline int to_bitmask() const {
|
simdjson_inline int to_bitmask() const {
|
||||||
|
|||||||
@@ -4,7 +4,7 @@
|
|||||||
#define SIMDJSON_SIMDJSON_VERSION_H
|
#define SIMDJSON_SIMDJSON_VERSION_H
|
||||||
|
|
||||||
/** The version of simdjson being used (major.minor.revision) */
|
/** The version of simdjson being used (major.minor.revision) */
|
||||||
#define SIMDJSON_VERSION "3.9.1"
|
#define SIMDJSON_VERSION "3.10.1"
|
||||||
|
|
||||||
namespace simdjson {
|
namespace simdjson {
|
||||||
enum {
|
enum {
|
||||||
@@ -15,7 +15,7 @@ enum {
|
|||||||
/**
|
/**
|
||||||
* The minor version (major.MINOR.revision) of simdjson being used.
|
* The minor version (major.MINOR.revision) of simdjson being used.
|
||||||
*/
|
*/
|
||||||
SIMDJSON_VERSION_MINOR = 9,
|
SIMDJSON_VERSION_MINOR = 10,
|
||||||
/**
|
/**
|
||||||
* The revision (major.minor.REVISION) of simdjson being used.
|
* The revision (major.minor.REVISION) of simdjson being used.
|
||||||
*/
|
*/
|
||||||
|
|||||||
@@ -35,13 +35,13 @@ static simdjson_inline uint32_t parse_eight_digits_unrolled(const uint8_t *chars
|
|||||||
simdjson_inline internal::value128 full_multiplication(uint64_t value1, uint64_t value2) {
|
simdjson_inline internal::value128 full_multiplication(uint64_t value1, uint64_t value2) {
|
||||||
internal::value128 answer;
|
internal::value128 answer;
|
||||||
#if SIMDJSON_REGULAR_VISUAL_STUDIO || SIMDJSON_IS_32BITS
|
#if SIMDJSON_REGULAR_VISUAL_STUDIO || SIMDJSON_IS_32BITS
|
||||||
#ifdef _M_ARM64
|
#if SIMDJSON_IS_ARM64
|
||||||
// ARM64 has native support for 64-bit multiplications, no need to emultate
|
// ARM64 has native support for 64-bit multiplications, no need to emultate
|
||||||
answer.high = __umulh(value1, value2);
|
answer.high = __umulh(value1, value2);
|
||||||
answer.low = value1 * value2;
|
answer.low = value1 * value2;
|
||||||
#else
|
#else
|
||||||
answer.low = _umul128(value1, value2, &answer.high); // _umul128 not available on ARM64
|
answer.low = _umul128(value1, value2, &answer.high); // _umul128 not available on ARM64
|
||||||
#endif // _M_ARM64
|
#endif // SIMDJSON_IS_ARM64
|
||||||
#else // SIMDJSON_REGULAR_VISUAL_STUDIO || SIMDJSON_IS_32BITS
|
#else // SIMDJSON_REGULAR_VISUAL_STUDIO || SIMDJSON_IS_32BITS
|
||||||
__uint128_t r = (static_cast<__uint128_t>(value1)) * value2;
|
__uint128_t r = (static_cast<__uint128_t>(value1)) * value2;
|
||||||
answer.low = uint64_t(r);
|
answer.low = uint64_t(r);
|
||||||
|
|||||||
@@ -59,10 +59,10 @@ namespace simd {
|
|||||||
struct simd8<bool>: base8<bool> {
|
struct simd8<bool>: base8<bool> {
|
||||||
static simdjson_inline simd8<bool> splat(bool _value) { return _mm_set1_epi8(uint8_t(-(!!_value))); }
|
static simdjson_inline simd8<bool> splat(bool _value) { return _mm_set1_epi8(uint8_t(-(!!_value))); }
|
||||||
|
|
||||||
simdjson_inline simd8<bool>() : base8() {}
|
simdjson_inline simd8() : base8() {}
|
||||||
simdjson_inline simd8<bool>(const __m128i _value) : base8<bool>(_value) {}
|
simdjson_inline simd8(const __m128i _value) : base8<bool>(_value) {}
|
||||||
// Splat constructor
|
// Splat constructor
|
||||||
simdjson_inline simd8<bool>(bool _value) : base8<bool>(splat(_value)) {}
|
simdjson_inline simd8(bool _value) : base8<bool>(splat(_value)) {}
|
||||||
|
|
||||||
simdjson_inline int to_bitmask() const { return _mm_movemask_epi8(*this); }
|
simdjson_inline int to_bitmask() const { return _mm_movemask_epi8(*this); }
|
||||||
simdjson_inline bool any() const { return !_mm_testz_si128(*this, *this); }
|
simdjson_inline bool any() const { return !_mm_testz_si128(*this, *this); }
|
||||||
|
|||||||
@@ -323,6 +323,10 @@ class Amalgamator:
|
|||||||
for line in fid2:
|
for line in fid2:
|
||||||
line = line.rstrip('\n')
|
line = line.rstrip('\n')
|
||||||
|
|
||||||
|
# Ignore #pragma once, it causes warnings if it ends up in a .cpp file
|
||||||
|
if re.search(r'^#pragma once$', line):
|
||||||
|
continue
|
||||||
|
|
||||||
# Ignore lines inside #ifndef SIMDJSON_CONDITIONAL_INCLUDE
|
# Ignore lines inside #ifndef SIMDJSON_CONDITIONAL_INCLUDE
|
||||||
if re.search(r'^#ifndef\s+SIMDJSON_CONDITIONAL_INCLUDE\s*$', line):
|
if re.search(r'^#ifndef\s+SIMDJSON_CONDITIONAL_INCLUDE\s*$', line):
|
||||||
assert file.is_conditional_include, f"{file} uses #ifndef SIMDJSON_CONDITIONAL_INCLUDE but is not an amalgamated file!"
|
assert file.is_conditional_include, f"{file} uses #ifndef SIMDJSON_CONDITIONAL_INCLUDE but is not an amalgamated file!"
|
||||||
|
|||||||
+247
-115
File diff suppressed because it is too large
Load Diff
+1725
-439
File diff suppressed because it is too large
Load Diff
@@ -263,7 +263,7 @@ simdjson_inline error_code json_structural_indexer::finish(dom_parser_implementa
|
|||||||
}
|
}
|
||||||
parser.n_structural_indexes = uint32_t(indexer.tail - parser.structural_indexes.get());
|
parser.n_structural_indexes = uint32_t(indexer.tail - parser.structural_indexes.get());
|
||||||
/***
|
/***
|
||||||
* The On Demand API requires special padding.
|
* The On-Demand API requires special padding.
|
||||||
*
|
*
|
||||||
* This is related to https://github.com/simdjson/simdjson/issues/906
|
* This is related to https://github.com/simdjson/simdjson/issues/906
|
||||||
* Basically, we want to make sure that if the parsing continues beyond the last (valid)
|
* Basically, we want to make sure that if the parsing continues beyond the last (valid)
|
||||||
|
|||||||
@@ -143,7 +143,7 @@ simdjson_inline bool handle_unicode_codepoint_wobbly(const uint8_t **src_ptr,
|
|||||||
* Unescape a valid UTF-8 string from src to dst, stopping at a final unescaped quote. There
|
* Unescape a valid UTF-8 string from src to dst, stopping at a final unescaped quote. There
|
||||||
* must be an unescaped quote terminating the string. It returns the final output
|
* must be an unescaped quote terminating the string. It returns the final output
|
||||||
* position as pointer. In case of error (e.g., the string has bad escaped codes),
|
* position as pointer. In case of error (e.g., the string has bad escaped codes),
|
||||||
* then null_nullptrptr is returned. It is assumed that the output buffer is large
|
* then null_ptr is returned. It is assumed that the output buffer is large
|
||||||
* enough. E.g., if src points at 'joe"', then dst needs to have four free bytes +
|
* enough. E.g., if src points at 'joe"', then dst needs to have four free bytes +
|
||||||
* SIMDJSON_PADDING bytes.
|
* SIMDJSON_PADDING bytes.
|
||||||
*/
|
*/
|
||||||
|
|||||||
@@ -7,6 +7,7 @@
|
|||||||
#include <internal/isadetection.h>
|
#include <internal/isadetection.h>
|
||||||
|
|
||||||
#include <initializer_list>
|
#include <initializer_list>
|
||||||
|
#include <type_traits>
|
||||||
|
|
||||||
namespace simdjson {
|
namespace simdjson {
|
||||||
|
|
||||||
@@ -169,8 +170,8 @@ namespace internal {
|
|||||||
*/
|
*/
|
||||||
class detect_best_supported_implementation_on_first_use final : public implementation {
|
class detect_best_supported_implementation_on_first_use final : public implementation {
|
||||||
public:
|
public:
|
||||||
const std::string &name() const noexcept final { return set_best()->name(); }
|
std::string name() const noexcept final { return set_best()->name(); }
|
||||||
const std::string &description() const noexcept final { return set_best()->description(); }
|
std::string description() const noexcept final { return set_best()->description(); }
|
||||||
uint32_t required_instruction_sets() const noexcept final { return set_best()->required_instruction_sets(); }
|
uint32_t required_instruction_sets() const noexcept final { return set_best()->required_instruction_sets(); }
|
||||||
simdjson_warn_unused error_code create_dom_parser_implementation(
|
simdjson_warn_unused error_code create_dom_parser_implementation(
|
||||||
size_t capacity,
|
size_t capacity,
|
||||||
@@ -190,6 +191,8 @@ private:
|
|||||||
const implementation *set_best() const noexcept;
|
const implementation *set_best() const noexcept;
|
||||||
};
|
};
|
||||||
|
|
||||||
|
static_assert(std::is_trivially_destructible<detect_best_supported_implementation_on_first_use>::value, "detect_best_supported_implementation_on_first_use should be trivially destructible");
|
||||||
|
|
||||||
static const std::initializer_list<const implementation *>& get_available_implementation_pointers() {
|
static const std::initializer_list<const implementation *>& get_available_implementation_pointers() {
|
||||||
static const std::initializer_list<const implementation *> available_implementation_pointers {
|
static const std::initializer_list<const implementation *> available_implementation_pointers {
|
||||||
#if SIMDJSON_IMPLEMENTATION_ICELAKE
|
#if SIMDJSON_IMPLEMENTATION_ICELAKE
|
||||||
@@ -246,6 +249,8 @@ public:
|
|||||||
unsupported_implementation() : implementation("unsupported", "Unsupported CPU (no detected SIMD instructions)", 0) {}
|
unsupported_implementation() : implementation("unsupported", "Unsupported CPU (no detected SIMD instructions)", 0) {}
|
||||||
};
|
};
|
||||||
|
|
||||||
|
static_assert(std::is_trivially_destructible<unsupported_implementation>::value, "unsupported_singleton should be trivially destructible");
|
||||||
|
|
||||||
const unsupported_implementation* get_unsupported_singleton() {
|
const unsupported_implementation* get_unsupported_singleton() {
|
||||||
static const unsupported_implementation unsupported_singleton{};
|
static const unsupported_implementation unsupported_singleton{};
|
||||||
return &unsupported_singleton;
|
return &unsupported_singleton;
|
||||||
|
|||||||
@@ -65,7 +65,7 @@ static inline uint32_t detect_supported_architectures() {
|
|||||||
return instruction_set::ALTIVEC;
|
return instruction_set::ALTIVEC;
|
||||||
}
|
}
|
||||||
|
|
||||||
#elif defined(__aarch64__) || defined(_M_ARM64)
|
#elif defined(__aarch64__) || defined(_M_ARM64) || defined(_M_ARM64EC)
|
||||||
|
|
||||||
static inline uint32_t detect_supported_architectures() {
|
static inline uint32_t detect_supported_architectures() {
|
||||||
return instruction_set::NEON;
|
return instruction_set::NEON;
|
||||||
|
|||||||
@@ -12,6 +12,7 @@ add_cpp_test(errortests LABELS dom acceptance per_implementation
|
|||||||
add_cpp_test(extracting_values_example LABELS dom acceptance per_implementation)
|
add_cpp_test(extracting_values_example LABELS dom acceptance per_implementation)
|
||||||
add_cpp_test(integer_tests LABELS dom acceptance per_implementation)
|
add_cpp_test(integer_tests LABELS dom acceptance per_implementation)
|
||||||
add_cpp_test(jsoncheck LABELS dom acceptance per_implementation)
|
add_cpp_test(jsoncheck LABELS dom acceptance per_implementation)
|
||||||
|
add_cpp_test(json_path_tests LABELS dom acceptance per_implementation)
|
||||||
add_cpp_test(minefieldcheck LABELS dom acceptance per_implementation)
|
add_cpp_test(minefieldcheck LABELS dom acceptance per_implementation)
|
||||||
add_cpp_test(numberparsingcheck LABELS dom acceptance per_implementation) # https://tools.ietf.org/html/rfc6901
|
add_cpp_test(numberparsingcheck LABELS dom acceptance per_implementation) # https://tools.ietf.org/html/rfc6901
|
||||||
add_cpp_test(parse_many_test LABELS dom acceptance per_implementation)
|
add_cpp_test(parse_many_test LABELS dom acceptance per_implementation)
|
||||||
|
|||||||
@@ -38,6 +38,25 @@ const size_t AMAZON_CELLPHONES_NDJSON_DOC_COUNT = 793;
|
|||||||
|
|
||||||
namespace number_tests {
|
namespace number_tests {
|
||||||
|
|
||||||
|
bool build(const std::string& json) {
|
||||||
|
simdjson::dom::parser parser;
|
||||||
|
simdjson::dom::element recvdJson;
|
||||||
|
auto error = parser.parse(json.c_str(), json.size()).get(recvdJson);
|
||||||
|
if (error) {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
bool issue2213() {
|
||||||
|
TEST_START();
|
||||||
|
std::string jsonStr = "[1,2,3,\"4\", {\"a\": 5}]";
|
||||||
|
for (int i = 0; i < 15; ++i) {
|
||||||
|
if (!build(jsonStr)) {
|
||||||
|
TEST_FAIL("The JSON is valid");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
TEST_SUCCEED();
|
||||||
|
}
|
||||||
bool ground_truth() {
|
bool ground_truth() {
|
||||||
std::cout << __func__ << std::endl;
|
std::cout << __func__ << std::endl;
|
||||||
std::pair<std::string,double> ground_truth[] = {
|
std::pair<std::string,double> ground_truth[] = {
|
||||||
@@ -397,7 +416,8 @@ namespace number_tests {
|
|||||||
}
|
}
|
||||||
|
|
||||||
bool run() {
|
bool run() {
|
||||||
return bomskip() &&
|
return issue2213() &&
|
||||||
|
bomskip() &&
|
||||||
issue2017() &&
|
issue2017() &&
|
||||||
truncated_borderline() &&
|
truncated_borderline() &&
|
||||||
specific_tests() &&
|
specific_tests() &&
|
||||||
@@ -950,6 +970,36 @@ namespace dom_api_tests {
|
|||||||
return true;
|
return true;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
bool convert_object_to_element() {
|
||||||
|
std::cout << "Running " << __func__ << std::endl;
|
||||||
|
string json(R"({ "a": 1, "b": 2, "c": 3 })");
|
||||||
|
|
||||||
|
dom::parser parser;
|
||||||
|
dom::object object;
|
||||||
|
dom::element element;
|
||||||
|
ASSERT_SUCCESS( parser.parse(json).get(object) );
|
||||||
|
element = object;
|
||||||
|
ASSERT_EQUAL( element["a"].get_uint64().value_unsafe(), 1 );
|
||||||
|
ASSERT_EQUAL( element["b"].get_uint64().value_unsafe(), 2 );
|
||||||
|
ASSERT_EQUAL( element["c"].get_uint64().value_unsafe(), 3 );
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
bool convert_array_to_element() {
|
||||||
|
std::cout << "Running " << __func__ << std::endl;
|
||||||
|
string json(R"([ 1, 10, 100 ])");
|
||||||
|
|
||||||
|
dom::parser parser;
|
||||||
|
dom::array array;
|
||||||
|
dom::element element;
|
||||||
|
ASSERT_SUCCESS( parser.parse(json).get(array) );
|
||||||
|
element = array;
|
||||||
|
ASSERT_EQUAL( element.at(0).get_uint64().value_unsafe(), 1 );
|
||||||
|
ASSERT_EQUAL( element.at(1).get_uint64().value_unsafe(), 10 );
|
||||||
|
ASSERT_EQUAL( element.at(2).get_uint64().value_unsafe(), 100 );
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
bool string_value() {
|
bool string_value() {
|
||||||
std::cout << "Running " << __func__ << std::endl;
|
std::cout << "Running " << __func__ << std::endl;
|
||||||
string json(R"([ "hi", "has backslash\\" ])");
|
string json(R"([ "hi", "has backslash\\" ])");
|
||||||
@@ -1331,6 +1381,8 @@ namespace dom_api_tests {
|
|||||||
array_iterator_empty() &&
|
array_iterator_empty() &&
|
||||||
object_iterator_advance() &&
|
object_iterator_advance() &&
|
||||||
array_iterator_advance() &&
|
array_iterator_advance() &&
|
||||||
|
convert_object_to_element() &&
|
||||||
|
convert_array_to_element() &&
|
||||||
string_value() &&
|
string_value() &&
|
||||||
numeric_values() &&
|
numeric_values() &&
|
||||||
boolean_values() &&
|
boolean_values() &&
|
||||||
|
|||||||
@@ -227,6 +227,49 @@ namespace document_stream_tests {
|
|||||||
return true;
|
return true;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
bool issue2170() {
|
||||||
|
TEST_START();
|
||||||
|
auto json = R"(1 2 3)"_padded;
|
||||||
|
simdjson::dom::parser parser;
|
||||||
|
simdjson::dom::document_stream stream;
|
||||||
|
ASSERT_SUCCESS(parser.parse_many(json).get(stream));
|
||||||
|
auto i = stream.begin();
|
||||||
|
size_t count{0};
|
||||||
|
std::vector<size_t> indexes = { 0, 2, 4 };
|
||||||
|
std::vector<std::string_view> expected = { "1", "2", "3" };
|
||||||
|
|
||||||
|
for(; i != stream.end(); ++i) {
|
||||||
|
auto doc = *i;
|
||||||
|
ASSERT_SUCCESS(doc);
|
||||||
|
ASSERT_TRUE(count < 3);
|
||||||
|
ASSERT_EQUAL(i.current_index(), indexes[count]);
|
||||||
|
ASSERT_EQUAL(i.source(), expected[count]);
|
||||||
|
count++;
|
||||||
|
}
|
||||||
|
TEST_SUCCEED();
|
||||||
|
}
|
||||||
|
|
||||||
|
bool issue2181() {
|
||||||
|
TEST_START();
|
||||||
|
auto json = R"(1 2 34)"_padded;
|
||||||
|
simdjson::dom::parser parser;
|
||||||
|
simdjson::dom::document_stream stream;
|
||||||
|
ASSERT_SUCCESS(parser.parse_many(json).get(stream));
|
||||||
|
auto i = stream.begin();
|
||||||
|
size_t count{0};
|
||||||
|
std::vector<size_t> indexes = { 0, 2, 4 };
|
||||||
|
std::vector<std::string_view> expected = { "1", "2", "34" };
|
||||||
|
|
||||||
|
for(; i != stream.end(); ++i) {
|
||||||
|
auto doc = *i;
|
||||||
|
ASSERT_SUCCESS(doc);
|
||||||
|
ASSERT_TRUE(count < 3);
|
||||||
|
ASSERT_EQUAL(i.current_index(), indexes[count]);
|
||||||
|
ASSERT_EQUAL(i.source(), expected[count]);
|
||||||
|
count++;
|
||||||
|
}
|
||||||
|
TEST_SUCCEED();
|
||||||
|
}
|
||||||
bool issue1310() {
|
bool issue1310() {
|
||||||
std::cout << "Running " << __func__ << std::endl;
|
std::cout << "Running " << __func__ << std::endl;
|
||||||
// hex : 20 20 5B 20 33 2C 31 5D 20 22 22 22 22 22 22 22 20 20 20 20 20 20 20 20 20 20 20 20 20 20 20 20 20
|
// hex : 20 20 5B 20 33 2C 31 5D 20 22 22 22 22 22 22 22 20 20 20 20 20 20 20 20 20 20 20 20 20 20 20 20 20
|
||||||
@@ -941,7 +984,9 @@ namespace document_stream_tests {
|
|||||||
}
|
}
|
||||||
|
|
||||||
bool run() {
|
bool run() {
|
||||||
return skipbom() &&
|
return issue2181() &&
|
||||||
|
issue2170() &&
|
||||||
|
skipbom() &&
|
||||||
fuzzaccess() &&
|
fuzzaccess() &&
|
||||||
baby_fuzzer() &&
|
baby_fuzzer() &&
|
||||||
issue1649() &&
|
issue1649() &&
|
||||||
|
|||||||
@@ -0,0 +1,350 @@
|
|||||||
|
/**
|
||||||
|
* refer to pathcheck.cpp
|
||||||
|
*/
|
||||||
|
|
||||||
|
#include <iostream>
|
||||||
|
#include <string>
|
||||||
|
using namespace std::string_literals;
|
||||||
|
|
||||||
|
#include "simdjson.h"
|
||||||
|
#include "test_macros.h"
|
||||||
|
|
||||||
|
// we define our own asserts to get around NDEBUG
|
||||||
|
#ifndef ASSERT
|
||||||
|
#define ASSERT(x) \
|
||||||
|
{ \
|
||||||
|
if (!(x)) { \
|
||||||
|
std::cerr << "Failed assertion " << #x << std::endl; \
|
||||||
|
return false; \
|
||||||
|
} \
|
||||||
|
}
|
||||||
|
#endif
|
||||||
|
|
||||||
|
using namespace simdjson;
|
||||||
|
|
||||||
|
bool demo() {
|
||||||
|
#if SIMDJSON_EXCEPTIONS
|
||||||
|
std::cout << "demo test" << std::endl;
|
||||||
|
auto cars_json = R"( [
|
||||||
|
{ "make": "Toyota", "model": "Camry", "year": 2018, "tire_pressure": [ 40.1, 39.9, 37.7, 40.4 ] },
|
||||||
|
{ "make": "Kia", "model": "Soul", "year": 2012, "tire_pressure": [ 30.1, 31.0, 28.6, 28.7 ] },
|
||||||
|
{ "make": "Toyota", "model": "Tercel", "year": 1999, "tire_pressure": [ 29.8, 30.0, 30.2, 30.5 ] }
|
||||||
|
] )"_padded;
|
||||||
|
dom::parser parser;
|
||||||
|
dom::element cars = parser.parse(cars_json);
|
||||||
|
double x = cars.at_path("$[0].tire_pressure[1]");
|
||||||
|
if (x != 39.9)
|
||||||
|
return false;
|
||||||
|
// Iterating through an array of objects
|
||||||
|
std::vector<double> measured;
|
||||||
|
for (dom::element car_element : cars) {
|
||||||
|
dom::object car;
|
||||||
|
simdjson::error_code error;
|
||||||
|
if ((error = car_element.get(car))) {
|
||||||
|
std::cerr << error << std::endl;
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
double x3 = car.at_path("$.tire_pressure[1]");
|
||||||
|
measured.push_back(x3);
|
||||||
|
}
|
||||||
|
std::vector<double> expected = {39.9, 31, 30};
|
||||||
|
if (measured != expected) {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
#endif
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
const padded_string TEST_JSON = R"(
|
||||||
|
{
|
||||||
|
"/~01abc": [
|
||||||
|
0,
|
||||||
|
{
|
||||||
|
"\\\" 0": [
|
||||||
|
"value0",
|
||||||
|
"value1"
|
||||||
|
]
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"0": "0 ok",
|
||||||
|
"01": "01 ok",
|
||||||
|
"": "empty ok",
|
||||||
|
"arr": []
|
||||||
|
}
|
||||||
|
)"_padded;
|
||||||
|
|
||||||
|
const padded_string TEST_RFC_JSON = R"(
|
||||||
|
{
|
||||||
|
"foo": ["bar", "baz"],
|
||||||
|
"": 0,
|
||||||
|
"a/b": 1,
|
||||||
|
"c%d": 2,
|
||||||
|
"e^f": 3,
|
||||||
|
"g|h": 4,
|
||||||
|
"i\\j": 5,
|
||||||
|
"k\"l": 6,
|
||||||
|
" ": 7,
|
||||||
|
"m~n": 8
|
||||||
|
}
|
||||||
|
)"_padded;
|
||||||
|
|
||||||
|
bool run_success_test(const padded_string &source, const char *json_path,
|
||||||
|
std::string_view expected_value) {
|
||||||
|
std::cout << "Running successful JSONPath test '" << json_path << "' ..."
|
||||||
|
<< std::endl;
|
||||||
|
dom::parser parser;
|
||||||
|
dom::element doc;
|
||||||
|
auto error = parser.parse(source).get(doc);
|
||||||
|
if (error) {
|
||||||
|
std::cerr << "cannot parse: " << error << std::endl;
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
dom::element answer;
|
||||||
|
error = doc.at_path(json_path).get(answer);
|
||||||
|
if (error) {
|
||||||
|
std::cerr << "cannot access pointer: " << error << std::endl;
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
std::string str_answer = simdjson::minify(answer);
|
||||||
|
if (str_answer != expected_value) {
|
||||||
|
std::cerr << "They differ!!!" << std::endl;
|
||||||
|
std::cerr << " found '" << str_answer << "'" << std::endl;
|
||||||
|
std::cerr << " expected '" << expected_value << "'" << std::endl;
|
||||||
|
}
|
||||||
|
ASSERT_EQUAL(str_answer, expected_value);
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
bool run_failure_test(const padded_string &source, const char *json_path,
|
||||||
|
error_code expected_error) {
|
||||||
|
std::cout << "Running invalid JSONPath test '" << json_path << "' ..."
|
||||||
|
<< std::endl;
|
||||||
|
dom::parser parser;
|
||||||
|
ASSERT_ERROR(parser.parse(source).at_path(json_path).error(), expected_error);
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
bool demo_relative_path() {
|
||||||
|
TEST_START();
|
||||||
|
auto cars_json = R"( [
|
||||||
|
{ "make": "Toyota", "model": "Camry", "year": 2018, "tire_pressure": [ 40.1, 39.9, 37.7, 40.4 ] },
|
||||||
|
{ "make": "Kia", "model": "Soul", "year": 2012, "tire_pressure": [ 30.1, 31.0, 28.6, 28.7 ] },
|
||||||
|
{ "make": "Toyota", "model": "Tercel", "year": 1999, "tire_pressure": [ 29.8, 30.0, 30.2, 30.5 ] }
|
||||||
|
] )"_padded;
|
||||||
|
|
||||||
|
dom::parser parser;
|
||||||
|
dom::element cars;
|
||||||
|
std::vector<double> measured;
|
||||||
|
auto error = parser.parse(cars_json).get(cars);
|
||||||
|
if (error) {
|
||||||
|
std::cerr << "cannot parse: " << error << std::endl;
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
dom::array cars_array;
|
||||||
|
error = cars.get(cars_array);
|
||||||
|
if (error) {
|
||||||
|
std::cerr << "cannot get array: " << error << std::endl;
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
for (auto car_element : cars_array) {
|
||||||
|
double x;
|
||||||
|
ASSERT_SUCCESS(car_element.at_path(".tire_pressure[1]").get(x));
|
||||||
|
measured.push_back(x);
|
||||||
|
}
|
||||||
|
|
||||||
|
std::vector<double> expected = {39.9, 31, 30};
|
||||||
|
if (measured != expected) {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
TEST_SUCCEED();
|
||||||
|
}
|
||||||
|
|
||||||
|
bool many_json_paths() {
|
||||||
|
TEST_START();
|
||||||
|
auto cars_json = R"( [
|
||||||
|
{ "make": "Toyota", "model": "Camry", "year": 2018, "tire_pressure": [ 40.1, 39.9, 37.7, 40.4 ] },
|
||||||
|
{ "make": "Kia", "model": "Soul", "year": 2012, "tire_pressure": [ 30.1, 31.0, 28.6, 28.7 ] },
|
||||||
|
{ "make": "Toyota", "model": "Tercel", "year": 1999, "tire_pressure": [ 29.8, 30.0, 30.2, 30.5 ] }
|
||||||
|
] )"_padded;
|
||||||
|
|
||||||
|
dom::parser parser;
|
||||||
|
dom::element cars;
|
||||||
|
std::vector<double> measured;
|
||||||
|
ASSERT_SUCCESS(parser.parse(cars_json).get(cars));
|
||||||
|
for (int i = 0; i < 3; i++) {
|
||||||
|
double x;
|
||||||
|
std::string json_path = std::string("$[") + std::to_string(i) +
|
||||||
|
std::string("].tire_pressure[1]");
|
||||||
|
ASSERT_SUCCESS(cars.at_path(json_path).get(x));
|
||||||
|
measured.push_back(x);
|
||||||
|
}
|
||||||
|
|
||||||
|
std::vector<double> expected = {39.9, 31, 30};
|
||||||
|
if (measured != expected) {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
TEST_SUCCEED();
|
||||||
|
}
|
||||||
|
|
||||||
|
bool many_json_paths_object_array() {
|
||||||
|
TEST_START();
|
||||||
|
auto dogcatpotato =
|
||||||
|
R"( { "dog" : [1,2,3], "cat" : [5, 6, 7], "potato" : [1234]})"_padded;
|
||||||
|
|
||||||
|
dom::parser parser;
|
||||||
|
dom::element doc;
|
||||||
|
ASSERT_SUCCESS(parser.parse(dogcatpotato).get(doc));
|
||||||
|
dom::object obj;
|
||||||
|
ASSERT_SUCCESS(doc.get_object().get(obj));
|
||||||
|
int64_t x;
|
||||||
|
ASSERT_SUCCESS(obj.at_path("$.dog[1]").get(x));
|
||||||
|
ASSERT_EQUAL(x, 2);
|
||||||
|
ASSERT_SUCCESS(obj.at_path("$.potato[0]").get(x));
|
||||||
|
ASSERT_EQUAL(x, 1234);
|
||||||
|
TEST_SUCCEED();
|
||||||
|
}
|
||||||
|
bool many_json_paths_object() {
|
||||||
|
TEST_START();
|
||||||
|
auto cfoofoo2 =
|
||||||
|
R"( { "c" :{ "foo": { "a": [ 10, 20, 30 ] }}, "d": { "foo2": { "a": [ 10, 20, 30 ] }} , "e": 120 })"_padded;
|
||||||
|
dom::parser parser;
|
||||||
|
dom::element doc;
|
||||||
|
ASSERT_SUCCESS(parser.parse(cfoofoo2).get(doc));
|
||||||
|
dom::object obj;
|
||||||
|
ASSERT_SUCCESS(doc.get_object().get(obj));
|
||||||
|
int64_t x;
|
||||||
|
ASSERT_SUCCESS(obj.at_path("$.c.foo.a[1]").get(x));
|
||||||
|
ASSERT_EQUAL(x, 20);
|
||||||
|
ASSERT_SUCCESS(obj.at_path("$.d.foo2.a.2").get(x));
|
||||||
|
ASSERT_EQUAL(x, 30);
|
||||||
|
ASSERT_SUCCESS(obj.at_path("$.e").get(x));
|
||||||
|
ASSERT_EQUAL(x, 120);
|
||||||
|
TEST_SUCCEED();
|
||||||
|
}
|
||||||
|
bool many_json_paths_array() {
|
||||||
|
TEST_START();
|
||||||
|
auto cfoofoo2 =
|
||||||
|
R"( [ 111, 2, 3, { "foo": { "a": [ 10, 20, 33 ] }}, { "foo2": { "a": [ 10, 20, 30 ] }}, 1001 ])"_padded;
|
||||||
|
dom::parser parser;
|
||||||
|
dom::element doc;
|
||||||
|
ASSERT_SUCCESS(parser.parse(cfoofoo2).get(doc));
|
||||||
|
dom::array arr;
|
||||||
|
ASSERT_SUCCESS(doc.get_array().get(arr));
|
||||||
|
int64_t x;
|
||||||
|
ASSERT_SUCCESS(arr.at_path("$[3].foo.a[1]").get(x));
|
||||||
|
ASSERT_EQUAL(x, 20);
|
||||||
|
TEST_SUCCEED();
|
||||||
|
}
|
||||||
|
struct car_type {
|
||||||
|
std::string make;
|
||||||
|
std::string model;
|
||||||
|
uint64_t year;
|
||||||
|
std::vector<double> tire_pressure;
|
||||||
|
car_type(std::string_view _make, std::string_view _model, uint64_t _year,
|
||||||
|
std::vector<double> &&_tire_pressure)
|
||||||
|
: make{_make}, model{_model}, year(_year), tire_pressure(_tire_pressure) {
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
bool json_path_invalidation() {
|
||||||
|
TEST_START();
|
||||||
|
auto cars_json = R"( [
|
||||||
|
{ "make": "Toyota", "model": "Camry", "year": 2018, "tire_pressure": [ 40.1, 39.9, 37.7, 40.4 ] },
|
||||||
|
{ "make": "Kia", "model": "Soul", "year": 2012, "tire_pressure": [ 30.1, 31.0, 28.6, 28.7 ] },
|
||||||
|
{ "make": "Toyota", "model": "Tercel", "year": 1999, "tire_pressure": [ 29.8, 30.0, 30.2, 30.5 ] }
|
||||||
|
] )"_padded;
|
||||||
|
|
||||||
|
dom::parser parser;
|
||||||
|
dom::element cars;
|
||||||
|
std::vector<double> measured;
|
||||||
|
ASSERT_SUCCESS(parser.parse(cars_json).get(cars));
|
||||||
|
std::vector<car_type> content;
|
||||||
|
for (int i = 0; i < 3; i++) {
|
||||||
|
dom::object obj;
|
||||||
|
std::string json_path =
|
||||||
|
std::string("$[") + std::to_string(i) + std::string("]");
|
||||||
|
// Each successive at_path call invalidates
|
||||||
|
// previously parsed values, strings, objects and array.
|
||||||
|
ASSERT_SUCCESS(cars.at_path(json_path).get(obj));
|
||||||
|
// We materialize the object.
|
||||||
|
std::string_view make;
|
||||||
|
ASSERT_SUCCESS(obj["make"].get(make));
|
||||||
|
std::string_view model;
|
||||||
|
ASSERT_SUCCESS(obj["model"].get(model));
|
||||||
|
uint64_t year;
|
||||||
|
ASSERT_SUCCESS(obj["year"].get(year));
|
||||||
|
// We materialize the array.
|
||||||
|
dom::array arr;
|
||||||
|
ASSERT_SUCCESS(obj["tire_pressure"].get(arr));
|
||||||
|
std::vector<double> values;
|
||||||
|
for (auto x : arr) {
|
||||||
|
double value_double;
|
||||||
|
ASSERT_SUCCESS(x.get(value_double));
|
||||||
|
values.push_back(value_double);
|
||||||
|
}
|
||||||
|
content.emplace_back(make, model, year, std::move(values));
|
||||||
|
}
|
||||||
|
std::string expected[] = {"Toyota", "Kia", "Toyota"};
|
||||||
|
int i = 0;
|
||||||
|
for (car_type c : content) {
|
||||||
|
std::cout << c.make << " " << c.model << " " << c.year << "\n";
|
||||||
|
ASSERT_EQUAL(expected[i++], c.make);
|
||||||
|
}
|
||||||
|
TEST_SUCCEED();
|
||||||
|
}
|
||||||
|
// for 0.5 version and following (standard compliant)
|
||||||
|
bool modern_support() {
|
||||||
|
#if SIMDJSON_EXCEPTIONS
|
||||||
|
std::cout << "modern test" << std::endl;
|
||||||
|
auto example_json = R"({"key": "value", "array": [0, 1, 2]})"_padded;
|
||||||
|
dom::parser parser;
|
||||||
|
dom::element example = parser.parse(example_json);
|
||||||
|
std::string_view value_str = example.at_path("$.key");
|
||||||
|
ASSERT_EQUAL(value_str, "value");
|
||||||
|
int64_t array0 = example.at_path("$.array[0]");
|
||||||
|
ASSERT_EQUAL(array0, 0);
|
||||||
|
array0 = example.at_path("$.array").at_path("$[0]");
|
||||||
|
ASSERT_EQUAL(array0, 0);
|
||||||
|
ASSERT_ERROR(example.at_path("$.no_such_key").error(), NO_SUCH_FIELD);
|
||||||
|
ASSERT_ERROR(example.at_path("$.array[9]").error(), INDEX_OUT_OF_BOUNDS);
|
||||||
|
ASSERT_ERROR(example.at_path("$.array.not_a_num").error(), INCORRECT_TYPE);
|
||||||
|
ASSERT_ERROR(example.at_path("$.array.").error(), INVALID_JSON_POINTER);
|
||||||
|
#endif
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
int main() {
|
||||||
|
if (true && demo() && modern_support() &&
|
||||||
|
run_success_test(TEST_RFC_JSON, "$.foo", "[\"bar\",\"baz\"]") &&
|
||||||
|
run_success_test(TEST_RFC_JSON, "$.foo[0]", "\"bar\"") &&
|
||||||
|
run_success_test(TEST_RFC_JSON, "$.", "0") &&
|
||||||
|
run_success_test(TEST_RFC_JSON, "$.a/b", "1") &&
|
||||||
|
run_success_test(TEST_RFC_JSON, "$.c%d", "2") &&
|
||||||
|
run_success_test(TEST_RFC_JSON, "$.e^f", "3") &&
|
||||||
|
run_success_test(TEST_RFC_JSON, "$.g|h", "4") &&
|
||||||
|
run_success_test(TEST_RFC_JSON, "$.i\\j", "5") &&
|
||||||
|
run_success_test(TEST_RFC_JSON, "$.k\"l", "6") &&
|
||||||
|
run_success_test(TEST_RFC_JSON, "$. ", "7") &&
|
||||||
|
run_success_test(TEST_RFC_JSON, "$.m~n", "8") &&
|
||||||
|
run_success_test(TEST_JSON, "$./~01abc",
|
||||||
|
R"([0,{"\\\" 0":["value0","value1"]}])") &&
|
||||||
|
run_success_test(TEST_JSON, "$./~01abc[1]",
|
||||||
|
R"({"\\\" 0":["value0","value1"]})") &&
|
||||||
|
run_success_test(TEST_JSON, "$./~01abc[1].\\\" 0",
|
||||||
|
R"(["value0","value1"])") &&
|
||||||
|
run_success_test(TEST_JSON, "$.arr", R"([])") // get array
|
||||||
|
&&
|
||||||
|
run_failure_test(TEST_JSON, R"($./~01abc[1].\\\" 0[2])", NO_SUCH_FIELD) &&
|
||||||
|
run_failure_test(TEST_JSON, "$.arr[0]", INDEX_OUT_OF_BOUNDS) &&
|
||||||
|
run_failure_test(TEST_JSON, "/~01abc", INVALID_JSON_POINTER) &&
|
||||||
|
run_failure_test(TEST_JSON, ".~1abc", NO_SUCH_FIELD) &&
|
||||||
|
run_failure_test(TEST_JSON, "./~01abc.01", INVALID_JSON_POINTER) &&
|
||||||
|
run_failure_test(TEST_JSON, "./~01abc.", INVALID_JSON_POINTER) &&
|
||||||
|
run_failure_test(TEST_JSON, "./~01abc.-", INDEX_OUT_OF_BOUNDS)) {
|
||||||
|
std::cout << "Success!" << std::endl;
|
||||||
|
return 0;
|
||||||
|
} else {
|
||||||
|
std::cerr << "Failed!" << std::endl;
|
||||||
|
return 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user