mirror of
https://github.com/simdjson/simdjson
synced 2026-06-08 17:27:07 +00:00
Compare commits
28 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 0aefa38507 | |||
| e341c8b438 | |||
| 0ac0a80e28 | |||
| 6b9117c029 | |||
| dd92151971 | |||
| 0679c247f4 | |||
| 615218a3ad | |||
| 4c1b0a41d8 | |||
| ef563a4b09 | |||
| 7a9ff93388 | |||
| fc61d7c7ba | |||
| d506af0a79 | |||
| ccf8694510 | |||
| 9b67497ed0 | |||
| 0336684df7 | |||
| c19320dd6e | |||
| 412a5680e8 | |||
| a05a56856d | |||
| 58173a6a1f | |||
| 1721032cfd | |||
| b73877f95e | |||
| 09723897e9 | |||
| 49e231b634 | |||
| acdbbab916 | |||
| 5090247c34 | |||
| 4180e05730 | |||
| feea2bce2c | |||
| 692f43cd84 |
@@ -25,6 +25,7 @@ CompileFlags:
|
||||
Diagnostics:
|
||||
Suppress:
|
||||
- pp_including_mainfile_in_preamble
|
||||
- unused-includes
|
||||
---
|
||||
# Amalgamated files that require or partly define an implementation
|
||||
If:
|
||||
|
||||
@@ -0,0 +1 @@
|
||||
blank_issues_enabled: false
|
||||
@@ -18,7 +18,7 @@ We do not make changes to simdjson without clearly identifiable benefits, which
|
||||
|
||||
Is your issue:
|
||||
|
||||
1. A bug report? If so, please point at a reproducible test. Indicate whether you are willing or able to provide a bug fix as a pull request.
|
||||
1. A bug report? If so, please point at a reproducible test. Indicate whether you are willing or able to provide a bug fix as a pull request. As a matter of policy, we do not consider a compiler warning to be a bug.
|
||||
|
||||
2. A build issue? If so, provide all possible details regarding your system configuration. If we cannot reproduce your issue, we cannot fix it.
|
||||
|
||||
|
||||
Vendored
+5
@@ -110,5 +110,10 @@
|
||||
"semaphore": "cpp",
|
||||
"stop_token": "cpp",
|
||||
"cfenv": "cpp"
|
||||
},
|
||||
"cmake.configureSettings": {
|
||||
"CMAKE_EXPORT_COMPILE_COMMANDS": "YES",
|
||||
"SIMDJSON_DEVELOPER_MODE": "ON",
|
||||
"SIMDJSON_SINGLEHEADER": "OFF"
|
||||
}
|
||||
}
|
||||
+19
-10
@@ -3,7 +3,7 @@ cmake_minimum_required(VERSION 3.14)
|
||||
project(
|
||||
simdjson
|
||||
# The version number is modified by tools/release.py
|
||||
VERSION 3.9.4
|
||||
VERSION 3.10.1
|
||||
DESCRIPTION "Parsing gigabytes of JSON per second"
|
||||
HOMEPAGE_URL "https://simdjson.org/"
|
||||
LANGUAGES CXX C
|
||||
@@ -20,10 +20,14 @@ string(
|
||||
# ---- Options, variables ----
|
||||
|
||||
# These version numbers are modified by tools/release.py
|
||||
set(SIMDJSON_LIB_VERSION "22.0.0" CACHE STRING "simdjson library version")
|
||||
set(SIMDJSON_LIB_SOVERSION "22" CACHE STRING "simdjson library soversion")
|
||||
set(SIMDJSON_LIB_VERSION "23.0.0" CACHE STRING "simdjson library version")
|
||||
set(SIMDJSON_LIB_SOVERSION "23" CACHE STRING "simdjson library soversion")
|
||||
|
||||
option(SIMDJSON_BUILD_STATIC_LIB "Build simdjson_static library along with simdjson" OFF)
|
||||
option(SIMDJSON_BUILD_STATIC_LIB "Build simdjson_static library along with simdjson (only makes sense if BUILD_SHARED_LIBS=ON)" OFF)
|
||||
if(SIMDJSON_BUILD_STATIC_LIB AND NOT BUILD_SHARED_LIBS)
|
||||
message(WARNING "SIMDJSON_BUILD_STATIC_LIB only makes sense if BUILD_SHARED_LIBS is set to ON")
|
||||
message(WARNING "You might be building and installing a two identical static libraries.")
|
||||
endif()
|
||||
|
||||
option(SIMDJSON_ENABLE_THREADS "Link with thread support" ON)
|
||||
|
||||
@@ -51,6 +55,7 @@ endif()
|
||||
if(is_top_project)
|
||||
option(SIMDJSON_DEVELOPER_MODE "Enable targets for developing simdjson" OFF)
|
||||
option(BUILD_SHARED_LIBS "Build simdjson as a shared library" OFF)
|
||||
option(SIMDJSON_SINGLEHEADER "Disable singleheader generation" ON)
|
||||
endif()
|
||||
|
||||
include(cmake/handle-deprecations.cmake)
|
||||
@@ -155,11 +160,13 @@ endif()
|
||||
include(CMakePackageConfigHelpers)
|
||||
include(GNUInstallDirs)
|
||||
|
||||
install(
|
||||
FILES singleheader/simdjson.h
|
||||
DESTINATION "${CMAKE_INSTALL_INCLUDEDIR}"
|
||||
COMPONENT simdjson_Development
|
||||
)
|
||||
if(SIMDJSON_SINGLEHEADER)
|
||||
install(
|
||||
FILES singleheader/simdjson.h
|
||||
DESTINATION "${CMAKE_INSTALL_INCLUDEDIR}"
|
||||
COMPONENT simdjson_Development
|
||||
)
|
||||
endif()
|
||||
|
||||
install(
|
||||
TARGETS simdjson
|
||||
@@ -203,6 +210,7 @@ if(SIMDJSON_BUILD_STATIC_LIB)
|
||||
TARGETS simdjson_static
|
||||
EXPORT simdjson_staticTargets
|
||||
ARCHIVE COMPONENT simdjson_Development
|
||||
INCLUDES DESTINATION "${CMAKE_INSTALL_INCLUDEDIR}"
|
||||
)
|
||||
install(
|
||||
EXPORT simdjson_staticTargets
|
||||
@@ -286,8 +294,9 @@ add_subdirectory(tools) ## This needs to be before tests because of cxxopts
|
||||
# most of the data has been moved to https://github.com/simdjson/simdjson-data
|
||||
add_subdirectory(jsonexamples)
|
||||
|
||||
|
||||
if(SIMDJSON_SINGLEHEADER)
|
||||
add_subdirectory(singleheader)
|
||||
endif()
|
||||
|
||||
|
||||
|
||||
|
||||
@@ -38,7 +38,7 @@ PROJECT_NAME = simdjson
|
||||
# could be handy for archiving the generated documentation or if some version
|
||||
# control system is used.
|
||||
|
||||
PROJECT_NUMBER = "3.9.4"
|
||||
PROJECT_NUMBER = "3.10.1"
|
||||
|
||||
# Using the PROJECT_BRIEF tag one can provide an optional one line description
|
||||
# for a project that appears at the top of each page and should give viewer a
|
||||
|
||||
+3
-3
@@ -88,8 +88,8 @@ simdjson's source structure, from the top level, looks like this:
|
||||
* simdjson/ondemand.h: the `simdjson::ondemand` namespace. Includes all public ondemand classes.
|
||||
* simdjson/builtin.h: the `simdjson::builtin` namespace. Aliased to the most universal implementation available.
|
||||
* simdjson/builtin/ondemand.h: the `simdjson::builtin::ondemand` namespace.
|
||||
* simdjson/arm64|fallback|haswell|icelake|ppc64|westmere/ondemand.h: the `simdjson::<implementation>::ondemand` namespace. on demand compiled for the specific implementation.
|
||||
* simdjson/generic/ondemand/*.h: individual on demand classes, generically written.
|
||||
* simdjson/arm64|fallback|haswell|icelake|ppc64|westmere/ondemand.h: the `simdjson::<implementation>::ondemand` namespace. On-Demand compiled for the specific implementation.
|
||||
* simdjson/generic/ondemand/*.h: individual On-Demand classes, generically written.
|
||||
* simdjson/generic/ondemand/dependencies.h: dependencies on common, non-implementation-specific simdjson classes. This will be included before including amalgamated.h.
|
||||
* simdjson/generic/ondemand/amalgamated.h: all generic ondemand classes for an implementation.
|
||||
* **src:** The source files for non-inlined functionality (e.g. the architecture-specific parser
|
||||
@@ -99,7 +99,7 @@ simdjson's source structure, from the top level, looks like this:
|
||||
* *.cpp: other misc. implementations, such as `simdjson::implementation` and the minifier.
|
||||
* arm64|fallback|haswell|icelake|ppc64|westmere.cpp: Architecture-specific parser implementations.
|
||||
* generic/*.h: `simdjson::<implementation>` namespace. Generic implementation of the parser, particularly the `dom_parser_implementation`.
|
||||
* generic/stage1/*.h: `simdjson::<implementation>::stage1` namespace. Generic implementation of the simd-heavy tokenizer/indexer pass of the simdjson parser. Used for the On Demand interface
|
||||
* generic/stage1/*.h: `simdjson::<implementation>::stage1` namespace. Generic implementation of the simd-heavy tokenizer/indexer pass of the simdjson parser. Used for the On-Demand interface
|
||||
* generic/stage2/*.h: `simdjson::<implementation>::stage2` namespace. Generic implementation of the tape creator, which consumes the index from stage 1 and actually parses numbers and string and such. Used for the DOM interface.
|
||||
|
||||
Other important files and directories:
|
||||
|
||||
@@ -46,6 +46,7 @@ Real-world usage
|
||||
- [Meta Velox](https://velox-lib.io)
|
||||
- [Google Pax](https://github.com/google/paxml)
|
||||
- [milvus](https://github.com/milvus-io/milvus)
|
||||
- [QuestDB](https://questdb.io/blog/questdb-release-8-0-3/)
|
||||
- [Clang Build Analyzer](https://github.com/aras-p/ClangBuildAnalyzer)
|
||||
- [Shopify HeapProfiler](https://github.com/Shopify/heap-profiler)
|
||||
- [StarRocks](https://github.com/StarRocks/starrocks)
|
||||
@@ -179,7 +180,7 @@ The simdjson library takes advantage of modern microarchitectures, parallelizing
|
||||
instructions, reducing branch misprediction, and reducing data dependency to take advantage of each
|
||||
CPU's multiple execution cores.
|
||||
|
||||
Our default front-end is called On Demand, and we wrote a paper about it:
|
||||
Our default front-end is called On-Demand, and we wrote a paper about it:
|
||||
|
||||
- John Keiser, Daniel Lemire, [On-Demand JSON: A Better Way to Parse Documents?](http://arxiv.org/abs/2312.17149), Software: Practice and Experience 54 (6), 2024.
|
||||
|
||||
@@ -200,8 +201,8 @@ For the video inclined, <br />
|
||||
Funding
|
||||
-------
|
||||
|
||||
The work is supported by the Natural Sciences and Engineering Research Council of Canada under grant
|
||||
number RGPIN-2017-03910.
|
||||
The work is supported by the Natural Sciences and Engineering Research Council of Canada under grants
|
||||
RGPIN-2017-03910 and RGPIN-2024-03787.
|
||||
|
||||
[license]: LICENSE
|
||||
[license img]: https://img.shields.io/badge/License-Apache%202-blue.svg
|
||||
|
||||
@@ -13,7 +13,7 @@ class OnDemand {
|
||||
public:
|
||||
OnDemand() {
|
||||
if(!displayed_implementation) {
|
||||
std::cout << "On Demand implementation: " << builtin_implementation()->name() << std::endl;
|
||||
std::cout << "On-Demand implementation: " << builtin_implementation()->name() << std::endl;
|
||||
displayed_implementation = true;
|
||||
}
|
||||
}
|
||||
|
||||
Vendored
+1
-1
@@ -155,6 +155,6 @@ if(SIMDJSON_CXXOPTS)
|
||||
set_off(CXXOPTS_BUILD_TESTS)
|
||||
set_off(CXXOPTS_ENABLE_INSTALL)
|
||||
|
||||
import_dependency(cxxopts jarro2783/cxxopts 794c975)
|
||||
import_dependency(cxxopts jarro2783/cxxopts 5965670)
|
||||
add_dependency(cxxopts)
|
||||
endif()
|
||||
|
||||
+38
-14
@@ -229,16 +229,16 @@ codepage, and they may call SetFileApisToOEM accordingly.
|
||||
Documents are iterators
|
||||
-----------------------
|
||||
|
||||
The simdjson library relies on an approach to parsing JSON that we call "On Demand".
|
||||
The simdjson library relies on an approach to parsing JSON that we call "On-Demand".
|
||||
A `document` is *not* a fully-parsed JSON value; rather, it is an **iterator** over the JSON text.
|
||||
This means that while you iterate an array, or search for a field in an object, it is actually
|
||||
walking through the original JSON text, merrily reading commas and colons and brackets to make sure
|
||||
you get where you are going. This is the key to On Demand's performance: since it's just an iterator,
|
||||
you get where you are going. This is the key to On-Demand's performance: since it's just an iterator,
|
||||
it lets you parse values as you use them. And particularly, it lets you *skip* values you do not want
|
||||
to use. On Demand is also ideally suited when you want to capture part of the document without parsing it
|
||||
to use. On-Demand is also ideally suited when you want to capture part of the document without parsing it
|
||||
immediately (e.g., see [General direct access to the raw JSON string](#general-direct-access-to-the-raw-json-string)).
|
||||
|
||||
We refer to "On Demand" as a front-end component since it is an interface between the
|
||||
We refer to "On-Demand" as a front-end component since it is an interface between the
|
||||
low-level parsing functions and the user. It hides much of the complexity of parsing JSON
|
||||
documents.
|
||||
|
||||
@@ -344,7 +344,7 @@ floating-point values followed by an integer.
|
||||
|
||||
We invite you to keep the following rules in mind:
|
||||
1. While you are accessing the document, the `document` instance should remain in scope: it is your "iterator" which keeps track of where you are in the JSON document. By design, there is one and only one `document` instance per JSON document.
|
||||
2. Because On Demand is really just an iterator, you must fully consume the current object or array before accessing a sibling object or array.
|
||||
2. Because On-Demand is really just an iterator, you must fully consume the current object or array before accessing a sibling object or array.
|
||||
3. Values can only be consumed once, you should get the values and store them if you plan to need them multiple times. You are expected to access the keys of an object just once. You are expected to go through the values of an array just once.
|
||||
|
||||
The simdjson library makes generous use of `std::string_view` instances. If you are unfamiliar
|
||||
@@ -357,7 +357,7 @@ support for users who avoid exceptions. See [the simdjson error handling documen
|
||||
|
||||
* **Validate What You Use:** When calling `iterate`, the document is quickly indexed. If it is
|
||||
not a valid Unicode (UTF-8) string or if there is an unclosed string, an error may be reported right away.
|
||||
However, it is not fully validated. On Demand only fully validates the values you use and the
|
||||
However, it is not fully validated. On-Demand only fully validates the values you use and the
|
||||
structure leading to it. It means that at every step as you traverse the document, you may encounter an error. You can handle errors either with exceptions or with error codes.
|
||||
* **Extracting Values:** You can cast a JSON element to a native type:
|
||||
`double(element)`. This works for `std::string_view`, double, uint64_t, int64_t, bool,
|
||||
@@ -407,7 +407,7 @@ support for users who avoid exceptions. See [the simdjson error handling documen
|
||||
* **Field Access:** To get the value of the "foo" field in an object, use `object["foo"]`. This will
|
||||
scan through the object looking for the field with the matching string, doing a character-by-character
|
||||
comparison. It may generate the error `simdjson::NO_SUCH_FIELD` if there is no such key in the object, it may throw an exception (see [Error handling](#error-handling)). For efficiency reason, you should avoid looking up the same field repeatedly: e.g., do
|
||||
not do `object["foo"]` followed by `object["foo"]` with the same `object` instance. For best performance, you should try to query the keys in the same order they appear in the document. If you need several keys and you cannot predict the order they will appear in, it is recommended to iterate through all keys `for(auto field : object) {...}`. Generally, you should not mix and match iterating through an object (`for(auto field : object) {...}`) and key accesses (`object["foo"]`): if you need to iterate through an object after a key access, you need to call `reset()` on the object. Keep in mind that On Demand does not buffer or save the result of the parsing: if you repeatedly access `object["foo"]`, then it must repeatedly seek the key and parse the content. The library does not provide a distinct function to check if a key is present, instead we recommend you attempt to access the key: e.g., by doing `ondemand::value val{}; if (!object["foo"].get(val)) {...}`, you have that `val` contains the requested value inside the if clause. It is your responsibility as a user to temporarily keep a reference to the value (`auto v = object["foo"]`), or to consume the content and store it in your own data structures. If you consume an
|
||||
not do `object["foo"]` followed by `object["foo"]` with the same `object` instance. For best performance, you should try to query the keys in the same order they appear in the document. If you need several keys and you cannot predict the order they will appear in, it is recommended to iterate through all keys `for(auto field : object) {...}`. Generally, you should not mix and match iterating through an object (`for(auto field : object) {...}`) and key accesses (`object["foo"]`): if you need to iterate through an object after a key access, you need to call `reset()` on the object. Whenever you call `reset()`, you need to keep in mind that though you can iterate over the array repeatedly, values should be consumedonly once (e.g., repeatedly calling `unescaped_key()` on the same key is forbidden). Keep in mind that On-Demand does not buffer or save the result of the parsing: if you repeatedly access `object["foo"]`, then it must repeatedly seek the key and parse the content. The library does not provide a distinct function to check if a key is present, instead we recommend you attempt to access the key: e.g., by doing `ondemand::value val{}; if (!object["foo"].get(val)) {...}`, you have that `val` contains the requested value inside the if clause. It is your responsibility as a user to temporarily keep a reference to the value (`auto v = object["foo"]`), or to consume the content and store it in your own data structures. If you consume an
|
||||
object twice: `std::string_view(object["foo"]` followed by `std::string_view(object["foo"]` then your code
|
||||
is in error. Furthermore, you can only consume one field at a time, on the same object. The
|
||||
value instance you get from `content["bids"]` becomes invalid when you call `content["asks"]`.
|
||||
@@ -1107,9 +1107,9 @@ If you find yourself needing only fast Unicode functions, consider using the sim
|
||||
JSON Pointer
|
||||
------------
|
||||
|
||||
The simdjson library also supports [JSON pointer](https://tools.ietf.org/html/rfc6901) through the `at_pointer()` method, letting you reach further down into the document in a single call. JSON pointer is supported by both the [DOM approach](https://github.com/simdjson/simdjson/blob/master/doc/dom.md#json-pointer) as well as the On Demand approach.
|
||||
The simdjson library also supports [JSON pointer](https://tools.ietf.org/html/rfc6901) through the `at_pointer()` method, letting you reach further down into the document in a single call. JSON pointer is supported by both the [DOM approach](https://github.com/simdjson/simdjson/blob/master/doc/dom.md#json-pointer) as well as the On-Demand approach.
|
||||
|
||||
**Note:** The On Demand implementation of JSON pointer relies on `find_field` which implies that it does not unescape keys when matching.
|
||||
**Note:** The On-Demand implementation of JSON pointer relies on `find_field` which implies that it does not unescape keys when matching.
|
||||
|
||||
Consider the following example:
|
||||
|
||||
@@ -1727,7 +1727,7 @@ before printout the data.
|
||||
}
|
||||
```
|
||||
|
||||
Performance note: the On Demand front-end does not materialize the parsed numbers and other values. If you are accessing everything twice, you may need to parse them twice. Thus the rewind functionality is best suited for cases where the first pass only scans the structure of the document.
|
||||
Performance note: the On-Demand front-end does not materialize the parsed numbers and other values. If you are accessing everything twice, you may need to parse them twice. Thus the rewind functionality is best suited for cases where the first pass only scans the structure of the document.
|
||||
|
||||
Both arrays and objects have a similar method `reset()`. It is similar
|
||||
to the document `rewind()` method, except that it does not rewind the
|
||||
@@ -1766,7 +1766,7 @@ for (auto doc : docs) {
|
||||
```
|
||||
|
||||
|
||||
Unlike `parser.iterate`, `parser.iterate_many` may parse "on demand" (lazily). That is, no parsing may have been done before you enter the loop
|
||||
Unlike `parser.iterate`, `parser.iterate_many` may parse "On-Demand" (lazily). That is, no parsing may have been done before you enter the loop
|
||||
`for (auto doc : docs) {` and you should expect the parser to only ever fully parse one JSON document at a time.
|
||||
|
||||
As with `parser.iterate`, when calling `parser.iterate_many(string)`, no copy is made of the provided string input. The provided memory buffer may be accessed each time a JSON document is parsed. Calling `parser.iterate_many(string)` on a temporary string buffer (e.g., `docs = parser.parse_many("[1,2,3]"_padded)`) is unsafe (and will not compile) because the `document_stream` instance needs access to the buffer to return the JSON documents.
|
||||
@@ -2297,7 +2297,7 @@ We built simdjson with thread safety in mind.
|
||||
The simdjson library is single-threaded except for [`iterate_many`](iterate_many.md) and [`parse_many`](parse_many.md) which may use secondary threads under their control when the library is compiled with thread support.
|
||||
|
||||
|
||||
We recommend using one `parser` object per thread. When using the On Demand front-end (our default), you should access the `document` instances in a single-threaded manner since it
|
||||
We recommend using one `parser` object per thread. When using the On-Demand front-end (our default), you should access the `document` instances in a single-threaded manner since it
|
||||
acts as an iterator (and is therefore not thread safe).
|
||||
|
||||
The CPU detection, which runs the first time parsing is attempted and switches to the fastest
|
||||
@@ -2623,12 +2623,37 @@ int main(void) {
|
||||
}
|
||||
```
|
||||
|
||||
|
||||
* Example 4: Value capture with `std::string_view` instances
|
||||
|
||||
```cpp
|
||||
void example() {
|
||||
ondemand::parser parser;
|
||||
const padded_string json = R"({ "parent": {"child1": {"name": "John"} , "child2": {"name": "Daniel"}} })"_padded;
|
||||
auto doc = parser.iterate(json);
|
||||
ondemand::object parent = doc["parent"];
|
||||
// parent owns the focus
|
||||
ondemand::object c1 = parent["child1"];
|
||||
// c1 owns the focus
|
||||
//
|
||||
std::string_view as1 = c1["name"];
|
||||
// We have that as1 == "John", as long as 'parser' and 'json' live
|
||||
// c2 attempts to grab the focus from parent but fails
|
||||
ondemand::object c2 = parent["child2"];
|
||||
// c2 owns the focus, at this point c1 is invalid
|
||||
std::string_view as2 = c2["name"];
|
||||
// We have that as2 == "Daniel", as long as 'parser' and 'json' live
|
||||
std::cout << as1 << " " << as2 << std::endl; // prints John Daniel
|
||||
}
|
||||
```
|
||||
|
||||
Performance tips
|
||||
--------
|
||||
|
||||
|
||||
- Read [our performance notes](performance.md) for advanced topics.
|
||||
- The On Demand front-end works best when doing a single pass over the input: avoid calling `count_elements`, `rewind` and similar methods.
|
||||
- To better understand the operation of your On-Demand parser, and whether it is performing as well as you think it should be, there is a logger feature built in to simdjson! To use it, define the pre-processor directive `SIMDJSON_VERBOSE_LOGGING` prior to including the `simdjson.h` header, which enables logging in simdjson. Run your code. It may generate a lot of logging output; adding printouts from your application that show each section may be helpful. The log's output will show step-by-step information on state, buffer pointer position, depth, and key retrieval status. Importantly, unless `SIMDJSON_VERBOSE_LOGGING` is defined, logging is entirely disabled and thus carries no overhead.
|
||||
- The On-Demand front-end works best when doing a single pass over the input: avoid calling `count_elements`, `rewind`, `reset` and similar methods.
|
||||
- If you are familiar with assembly language, you may use the online tool godbolt to explore the compiled code. The following example may work: [https://godbolt.org/z/xE4GWs573](https://godbolt.org/z/xE4GWs573).
|
||||
- Given a field `field` in an object, calling `field.key()` is often faster than `field.unescaped_key()` so if you do not need an unescaped `std::string_view` instance, prefer `field.key()`. Similarly, we expect `field.escaped_key()` to be faster than `field.unescaped_key()` even though both return a `std::string_view` instance.
|
||||
- For release builds, we recommend setting `NDEBUG` pre-processor directive when compiling the `simdjson` library. Importantly, using the optimization flags `-O2` or `-O3` under GCC and LLVM clang does not set the `NDEBUG` directive, you must set it manually (e.g., `-DNDEBUG`).
|
||||
@@ -2648,7 +2673,6 @@ Performance tips
|
||||
std::string_view year = data["year"];
|
||||
std::string_view rating = data["rating"];
|
||||
```
|
||||
- To better understand the operation of your On Demand parser, and whether it is performing as well as you think it should be, there is a logger feature built in to simdjson! To use it, define the pre-processor directive `SIMDJSON_VERBOSE_LOGGING` prior to including the `simdjson.h` header, which enables logging in simdjson. Run your code. It may generate a lot of logging output; adding printouts from your application that show each section may be helpful. The log's output will show step-by-step information on state, buffer pointer position, depth, and key retrieval status. Importantly, unless `SIMDJSON_VERBOSE_LOGGING` is defined, logging is entirely disabled and thus carries no overhead.
|
||||
|
||||
|
||||
|
||||
|
||||
+2
-2
@@ -3,7 +3,7 @@ The Document-Object-Model (DOM) front-end
|
||||
|
||||
An overview of what you need to know to use simdjson, with examples.
|
||||
|
||||
* [DOM vs On Demand](#dom-vs-on-demand)
|
||||
* [DOM vs On-Demand](#dom-vs-on-demand)
|
||||
* [The Basics: Loading and Parsing JSON Documents](#the-basics-loading-and-parsing-json-documents-using-the-dom-front-end)
|
||||
* [Using the Parsed JSON](#using-the-parsed-json)
|
||||
* [C++17 Support](#c17-support)
|
||||
@@ -18,7 +18,7 @@ An overview of what you need to know to use simdjson, with examples.
|
||||
* [Padding and Temporary Copies](#padding-and-temporary-copies)
|
||||
* [Performance Tips](#performance-tips)
|
||||
|
||||
DOM vs On Demand
|
||||
DOM vs On-Demand
|
||||
----------------------------------------------
|
||||
|
||||
The simdjson library offers two distinct approaches on how to access a JSON document. We support
|
||||
|
||||
+30
-30
@@ -9,8 +9,8 @@ Whether we parse JSON or XML, or any other serialized format, there are relative
|
||||
- Another popular approach is the schema-based deserialization model.
|
||||
|
||||
We propose an approach that is as easy to use and often as flexible as the DOM approach, yet as fast and
|
||||
efficient as the schema-based or event-based approaches. We call this new approach "On Demand". The
|
||||
simdjson On Demand API offers a familiar, friendly DOM API and
|
||||
efficient as the schema-based or event-based approaches. We call this new approach "On-Demand". The
|
||||
simdjson On-Demand API offers a familiar, friendly DOM API and
|
||||
provides the performance of just-in-time parsing on top of the simdjson superior performance.
|
||||
|
||||
To achieve ease of use, we mimicked the *form* of a traditional DOM API: you can iterate over
|
||||
@@ -18,7 +18,7 @@ arrays, look up fields in objects, and extract native values like `double`, `uin
|
||||
|
||||
To achieve performance, we introduced some key limitations that make the DOM API *streaming*:
|
||||
array/object iteration cannot be restarted, and string/number values can only be parsed once. If
|
||||
these limitations are acceptable to you, the On Demand API could help you write maintainable
|
||||
these limitations are acceptable to you, the On-Demand API could help you write maintainable
|
||||
applications with a computation efficiency that is difficult to surpass.
|
||||
|
||||
A code example illustrates our API from a programmer's point of view:
|
||||
@@ -72,24 +72,24 @@ This streaming approach means that unused fields and values are not parsed or
|
||||
converted, thus saving space and time. In our example, the `"name"`, `"followers_count"`,
|
||||
and `"friends_count"` keys and matching values are skipped.
|
||||
|
||||
Further, the On Demand API does not parse a value *at all* until you try to convert it (e.g., to `double`,
|
||||
Further, the On-Demand API does not parse a value *at all* until you try to convert it (e.g., to `double`,
|
||||
`int`, `string`, or `bool`). In our example, when accessing the key-value pair `"retweet_count": 82`, the parser
|
||||
may not convert the pair of characters `82` to the binary integer 82. Because the programmer specifies the data
|
||||
type, we avoid branch mispredictions related to data type determination and improve the performance.
|
||||
|
||||
|
||||
We expect users of an On Demand API to work in terms of a JSON dialect, which is a set of expectations and
|
||||
We expect users of an On-Demand API to work in terms of a JSON dialect, which is a set of expectations and
|
||||
specifications that come in addition to the [JSON specification](https://www.rfc-editor.org/rfc/rfc8259.txt).
|
||||
The On Demand approach is designed around several principles:
|
||||
The On-Demand approach is designed around several principles:
|
||||
|
||||
* **Streaming (\*):** It avoids preparsing values, keeping the memory usage and the latency down.
|
||||
* **Forward-Only:** To prevent reiteration of the same values and to keep the number of variables down (literally), only a single index is maintained and everything uses it (even if you have nested for loops). This means when you are going through an array of arrays, for example, that the inner array loop will advance the index to the next comma, and the array can just pick it up and look at it.
|
||||
* **Natural Iteration:** A JSON array or object can be iterated with a normal C++ for loop. Nested arrays and objects are supported by nested for loops.
|
||||
* **Use-Specific Parsing:** Parsing is always specific to the type required by the programmer. For example, if the programmer asks for an unsigned integer, we just start parsing digits. If there were no digits, we toss an error. There are even different parsers for `double`, `uint64_t` and `int64_t` values. This use-specific parsing avoids the branchiness of a generic "type switch," and makes the code more inlineable and compact.
|
||||
* **Validate What You Use:** On Demand deliberately validates the values you use and the structure leading to it, but nothing else. The goal is a guarantee that the value you asked for is the correct one and is not malformed: there must be no confusion over whether you got the right value.
|
||||
* **Validate What You Use:** On-Demand deliberately validates the values you use and the structure leading to it, but nothing else. The goal is a guarantee that the value you asked for is the correct one and is not malformed: there must be no confusion over whether you got the right value.
|
||||
|
||||
|
||||
To understand why On Demand is different, it is helpful to review the major
|
||||
To understand why On-Demand is different, it is helpful to review the major
|
||||
approaches to parsing and parser APIs in use today.
|
||||
|
||||
### DOM Parsers
|
||||
@@ -106,7 +106,7 @@ DOM tree is often easy enough that many users use the DOM as-is instead of creat
|
||||
their own custom data structures.
|
||||
|
||||
The DOM approach was the only way to parse JSON documents up to version 0.6 of the simdjson library.
|
||||
Our DOM API looks similar to our On Demand example, except
|
||||
Our DOM API looks similar to our On-Demand example, except
|
||||
it calls `parse` instead of `iterate`:
|
||||
|
||||
```c++
|
||||
@@ -152,7 +152,7 @@ a tweet right now, or is this from some other place in the document
|
||||
entirely? Though an event-based approach may allow superior performance, it is demanding of the programmer
|
||||
who must efficiently keep track of its current state within the JSON input.
|
||||
|
||||
The following is event-based example of the Twitter problem we have reviewed in the DOM and On Demand
|
||||
The following is event-based example of the Twitter problem we have reviewed in the DOM and On-Demand
|
||||
examples. To make it short enough to use as an example at all, it has heavily redacted: it only solves
|
||||
a part of the problem (does not get user.screen_name), it has bugs (it does not handle sub-objects
|
||||
in a tweet at all), and it uses a theoretical, simple event-based API that minimizes ceremony.
|
||||
@@ -257,7 +257,7 @@ stress the branch prediction. Though branch predictors improve with each new gen
|
||||
the cost of branch mispredictions also tends to increase as pipelines expand, and the processors become
|
||||
able to schedule longer streams of instructions.
|
||||
|
||||
On Demand parsing is tailor-made to solve this problem at the source, parsing values only after the
|
||||
On-Demand parsing is tailor-made to solve this problem at the source, parsing values only after the
|
||||
user declares their type by asking for a `double`, an `int`, a `string`, etc. It attempts to do so while
|
||||
preserving most of the flexibility of DOM parsing.
|
||||
|
||||
@@ -297,7 +297,7 @@ To help visualize the algorithm, we'll walk through the example C++ given at the
|
||||
|
||||
Since this is the first time this parser has been used, `iterate()` first allocates internal
|
||||
parser buffers if this is the first time through. When reusing an existing parser, allocation
|
||||
only happens if the new document is bigger than internal buffers can handle. The On Demand
|
||||
only happens if the new document is bigger than internal buffers can handle. The On-Demand
|
||||
API only ever allocates memory in the `iterate()` function call.
|
||||
|
||||
The simdjson library then preprocesses the JSON text at high speed, finding all tokens (i.e. the starting
|
||||
@@ -492,7 +492,7 @@ To help visualize the algorithm, we'll walk through the example C++ given at the
|
||||
Because of the cast to uint64_t, simdjson knows it's parsing an unsigned integer. This lets
|
||||
us use a fast parser which *only* knows how to parse digits. It validates that it is an integer
|
||||
by rejecting negative numbers, strings, and other values based on the fact that they are not the
|
||||
digits 0-9. This type specificity is part of why parsing with on demand is so fast: you lose all
|
||||
digits 0-9. This type specificity is part of why parsing with On-Demand is so fast: you lose all
|
||||
the code that has to understand those other types.
|
||||
|
||||
The iterator is advanced to the `}`, and depth decreased back to 3 (root > statuses > tweet).
|
||||
@@ -597,7 +597,7 @@ To help visualize the algorithm, we'll walk through the example C++ given at the
|
||||
|
||||
This means you can very efficiently do things like read a single value from a JSON file, or take
|
||||
the top N, for example. It also means the things you don't use won't be fully validated. This is
|
||||
a general principle of On Demand: don't validate what you don't use. We still fully validate
|
||||
a general principle of On-Demand: don't validate what you don't use. We still fully validate
|
||||
values you do use, however, as well as the objects and arrays that lead to them, so that you can
|
||||
be sure you get the information you need.
|
||||
|
||||
@@ -654,7 +654,7 @@ for(auto field : doc.get_object()) {
|
||||
|
||||
### Iteration Safety
|
||||
|
||||
The On Demand API is powerful. To compensate, we add some safeguards to ensure that it can be used without fear
|
||||
The On-Demand API is powerful. To compensate, we add some safeguards to ensure that it can be used without fear
|
||||
in production systems:
|
||||
|
||||
- If the value fails to be parsed as one type, the program can try to parse it as something else until the program succeeds. Thus
|
||||
@@ -667,7 +667,7 @@ in production systems:
|
||||
if it was `nullptr` but did not care what the actual value was--it will iterate. The destructor automates
|
||||
the iteration.
|
||||
|
||||
Some care is needed when using the On Demand API in scenarios where you need to access several sibling arrays or objects because
|
||||
Some care is needed when using the On-Demand API in scenarios where you need to access several sibling arrays or objects because
|
||||
only one object or array can be active at any one time. Let us consider the following example:
|
||||
|
||||
```C++
|
||||
@@ -709,36 +709,36 @@ A correct usage is given by the following example:
|
||||
}
|
||||
```
|
||||
|
||||
### Benefits of the On Demand Approach
|
||||
### Benefits of the On-Demand Approach
|
||||
|
||||
We expect that the On Demand approach has many of the performance benefits of the schema-based approach, while providing a flexibility that is similar to that of the DOM-based approach.
|
||||
We expect that the On-Demand approach has many of the performance benefits of the schema-based approach, while providing a flexibility that is similar to that of the DOM-based approach.
|
||||
|
||||
* Faster than DOM in some cases. Reduced memory usage.
|
||||
* Straightforward, programmer-friendly interface (arrays and objects).
|
||||
* Highly expressive, beyond deserialization and pointer queries: many tasks can be accomplished with little code.
|
||||
|
||||
### Limitations of the On Demand Approach
|
||||
### Limitations of the On-Demand Approach
|
||||
|
||||
The On Demand approach has some limitations:
|
||||
The On-Demand approach has some limitations:
|
||||
|
||||
* Because it operates in streaming mode, you only have access to the current element in the JSON document. Furthermore, the document is traversed in order so the code is sensitive to the order of the JSON nodes in the same manner as an event-based approach (e.g., SAX). (The one exception to this is field lookup, which is more *performant* when the order of lookups matches the order of fields in the document, but which will still work with out-of-order fields, with a performance hit.)
|
||||
* The On Demand approach is less safe than DOM: we only validate the components of the JSON document that are used and it is possible to begin ingesting an invalid document only to find out later that the document is invalid. Are you fine ingesting a large JSON document that starts with well formed JSON but ends with invalid JSON content?
|
||||
* The On-Demand approach is less safe than DOM: we only validate the components of the JSON document that are used and it is possible to begin ingesting an invalid document only to find out later that the document is invalid. Are you fine ingesting a large JSON document that starts with well formed JSON but ends with invalid JSON content?
|
||||
|
||||
There are currently additional technical limitations which we expect to resolve in future releases of the simdjson library:
|
||||
|
||||
* The simdjson library offers runtime dispatching which allows you to compile one binary and have it run at full speed on different processors, taking advantage of the specific features of the processor. The On Demand API has limited runtime dispatch support. Under x64 systems, to fully benefit from the On Demand API, we recommend that you compile your code for a specific processor. E.g., if your processor supports AVX2 instructions, you should compile your binary executable with AVX2 instruction support (by using your compiler's commands). If you are sufficiently technically proficient, you can implement runtime dispatching within your application, by compiling your On Demand code for different processors.
|
||||
* The simdjson library offers runtime dispatching which allows you to compile one binary and have it run at full speed on different processors, taking advantage of the specific features of the processor. The On-Demand API has limited runtime dispatch support. Under x64 systems, to fully benefit from the On-Demand API, we recommend that you compile your code for a specific processor. E.g., if your processor supports AVX2 instructions, you should compile your binary executable with AVX2 instruction support (by using your compiler's commands). If you are sufficiently technically proficient, you can implement runtime dispatching within your application, by compiling your On-Demand code for different processors.
|
||||
* There is an initial phase which scans the entire document quickly, irrespective of the size of the document. We plan to break this phase into distinct steps for large files in a future release as we have done with other components of our API (e.g., `parse_many`).
|
||||
|
||||
### Applicability of the On Demand Approach
|
||||
### Applicability of the On-Demand Approach
|
||||
|
||||
At this time we recommend the On Demand API in the following cases:
|
||||
At this time we recommend the On-Demand API in the following cases:
|
||||
|
||||
1. The 64-bit hardware (CPU) used to run the software is known at compile time. If you need runtime dispatching because you cannot be certain of the hardware used to run your software, you will be better served with the core simdjson API. (This only applies to x64 (AMD/Intel). On 64-bit ARM hardware, runtime dispatching is unnecessary.)
|
||||
2. The used parts of JSON files do not need to be validated and the layout of the nodes follows a strict JSON dialect. If you are receiving JSON from other systems, you might be better served with core simdjson API as it fully validates the JSON inputs and allows you to navigate through the document at will.
|
||||
3. Speed and efficiency are of the utmost importance. Keep in mind that the core simdjson API is highly efficient so adopting the On Demand API is not necessary for high efficiency.
|
||||
3. Speed and efficiency are of the utmost importance. Keep in mind that the core simdjson API is highly efficient so adopting the On-Demand API is not necessary for high efficiency.
|
||||
4. As a developer, you value a clean, flexible and maintainable API.
|
||||
|
||||
Good applications for the On Demand API might be:
|
||||
Good applications for the On-Demand API might be:
|
||||
|
||||
* You are working from pre-existing large JSON files that have been vetted. You expect them to be well formed according to a known JSON dialect and to have a consistent layout. For example, you might be doing biomedical research or machine learning on top of static data dumps in JSON.
|
||||
* Both the generation and the consumption of JSON data is within your system. Your team controls both the software that produces the JSON and the software the parses it, your team knows and control the hardware. Thus you can fully test your system.
|
||||
@@ -746,13 +746,13 @@ Good applications for the On Demand API might be:
|
||||
|
||||
## Checking Your CPU Selection (x64 systems)
|
||||
|
||||
The On Demand API uses advanced architecture-specific code for many common processors to make JSON preprocessing and string parsing faster. By default, however, most c++ compilers will compile to the least common denominator (since the program could theoretically be run anywhere). Since On Demand is inlined into your own code, it cannot always use these advanced versions unless the compiler is told to target them.
|
||||
The On-Demand API uses advanced architecture-specific code for many common processors to make JSON preprocessing and string parsing faster. By default, however, most c++ compilers will compile to the least common denominator (since the program could theoretically be run anywhere). Since On-Demand is inlined into your own code, it cannot always use these advanced versions unless the compiler is told to target them.
|
||||
|
||||
On relevant systems, the On Demand API provides some support for runtime dispatching: that is, it will attempt to detect, at runtime, the instructions that your processor supports and optimize the code accordingly. However, it cannot always make full use of the features of your processor.
|
||||
On relevant systems, the On-Demand API provides some support for runtime dispatching: that is, it will attempt to detect, at runtime, the instructions that your processor supports and optimize the code accordingly. However, it cannot always make full use of the features of your processor.
|
||||
|
||||
Some users wish to run at the best possible speed. Under recent Intel and AMD processors, these users should take additional steps to verify that their code is well optimized.
|
||||
|
||||
Given that the On Demand API offer limited runtime dispatching, it matters that your code is compiled against a specific CPU target. You should verify that the code is compiled against the target you expect. Thankfully, the simdjson library will tell you exactly what it detects as an implementation: `icelake` (AVX512 x64 processors), `haswell` (AVX2 x64 processors), `westmere` (SSE4 x64 processors), `arm64` (64-bit ARM), `ppc64` (64-bit POWER), `lasx` (LoongArch), `lsx` (LoongArch), `fallback` (others). Under x64 processors, many programmers will want to target `haswell` whereas under ARM, most programmers will want to target `arm64` (and it should do so automatically). The `fallback` is probably only good for testing purposes, not for deployment.
|
||||
Given that the On-Demand API offer limited runtime dispatching, it matters that your code is compiled against a specific CPU target. You should verify that the code is compiled against the target you expect. Thankfully, the simdjson library will tell you exactly what it detects as an implementation: `icelake` (AVX512 x64 processors), `haswell` (AVX2 x64 processors), `westmere` (SSE4 x64 processors), `arm64` (64-bit ARM), `ppc64` (64-bit POWER), `lasx` (LoongArch), `lsx` (LoongArch), `fallback` (others). Under x64 processors, many programmers will want to target `haswell` whereas under ARM, most programmers will want to target `arm64` (and it should do so automatically). The `fallback` is probably only good for testing purposes, not for deployment.
|
||||
|
||||
```C++
|
||||
std::cout << simdjson::builtin_implementation()->name() << std::endl;
|
||||
@@ -777,6 +777,6 @@ In these examples, the `-march=haswell` flags targets a haswell processor and th
|
||||
|
||||
Instead of specifying a specific microarchitecture, you can let your compiler do the work. The `-march=native` flags says "target the current computer," which is a reasonable default for many applications which both compile and run on the same processor.
|
||||
|
||||
Passing `-march=native` to the compiler may make On Demand faster by allowing it to use optimizations specific to your machine. You cannot do this, however, if you are compiling code that might be run on less advanced machines. That is, be mindful that when compiling with the `-march=native` flag, the resulting binary will run on the current system but may not run on other systems (e.g., on an old processor).
|
||||
Passing `-march=native` to the compiler may make On-Demand faster by allowing it to use optimizations specific to your machine. You cannot do this, however, if you are compiling code that might be run on less advanced machines. That is, be mindful that when compiling with the `-march=native` flag, the resulting binary will run on the current system but may not run on other systems (e.g., on an old processor).
|
||||
|
||||
If you are compiling on an ARM or POWER system, you do not need to be concerned with CPU selection during compilation. The `-march=native` flag is useful for best performance on x64 (e.g., Intel) systems but it is generally unsupported on some platforms such as ARM (aarch64) or POWER.
|
||||
|
||||
+1
-1
@@ -125,7 +125,7 @@ Whitespace Characters:
|
||||
- **Nothing**
|
||||
|
||||
Some official formats **(non-exhaustive list)**:
|
||||
- [Newline-Delimited JSON (NDJSON)](http://ndjson.org/)
|
||||
- [Newline-Delimited JSON (NDJSON)](https://github.com/ndjson/ndjson-spec)
|
||||
- [JSON lines (JSONL)](http://jsonlines.org/)
|
||||
- [Record separator-delimited JSON (RFC 7464)](https://tools.ietf.org/html/rfc7464) <- Not supported by JsonStream!
|
||||
- [More on Wikipedia...](https://en.wikipedia.org/wiki/JSON_streaming)
|
||||
|
||||
+1
-1
@@ -75,7 +75,7 @@ or simply
|
||||
Server Loops: Long-Running Processes and Memory Capacity
|
||||
---------------------------------
|
||||
|
||||
The On Demand approach also automatically expands its memory capacity when larger documents are parsed. However, for longer processes where very large files are processed (such as server loops), this capacity is not resized down. On Demand also lets you adjust the maximal capacity that the parser can process:
|
||||
The On-Demand approach also automatically expands its memory capacity when larger documents are parsed. However, for longer processes where very large files are processed (such as server loops), this capacity is not resized down. On-Demand also lets you adjust the maximal capacity that the parser can process:
|
||||
|
||||
* You can set an upper bound (*max_capacity*) when construction the parser:
|
||||
```C++
|
||||
|
||||
@@ -25,7 +25,7 @@ IF(${CMAKE_SYSTEM_NAME} MATCHES "Linux")
|
||||
add_quickstart_test(quickstart2_noexceptions quickstart2_noexceptions.cpp NO_EXCEPTIONS LABELS acceptance)
|
||||
add_quickstart_test(quickstart2_noexceptions11 quickstart2_noexceptions.cpp NO_EXCEPTIONS CXX_STANDARD c++11)
|
||||
|
||||
# On Demand Quick Start
|
||||
# On-Demand Quick Start
|
||||
if (SIMDJSON_EXCEPTIONS)
|
||||
add_quickstart_test(quickstart_ondemand quickstart_ondemand.cpp LABELS quickstart_ondemand acceptance)
|
||||
add_quickstart_test(quickstart_ondemand11 quickstart_ondemand.cpp CXX_STANDARD c++11 LABELS quickstart_ondemand acceptance)
|
||||
|
||||
@@ -50,6 +50,8 @@ double from_chars(const char *first, const char* end) noexcept;
|
||||
#define SIMDJSON_ISALIGNED_N(ptr, n) (((uintptr_t)(ptr) & ((n)-1)) == 0)
|
||||
|
||||
#if SIMDJSON_REGULAR_VISUAL_STUDIO
|
||||
// We could use [[deprecated]] but it requires C++14
|
||||
#define simdjson_deprecated __declspec(deprecated)
|
||||
|
||||
#define simdjson_really_inline __forceinline
|
||||
#define simdjson_never_inline __declspec(noinline)
|
||||
@@ -88,6 +90,8 @@ double from_chars(const char *first, const char* end) noexcept;
|
||||
#define SIMDJSON_POP_DISABLE_UNUSED_WARNINGS
|
||||
|
||||
#else // SIMDJSON_REGULAR_VISUAL_STUDIO
|
||||
// We could use [[deprecated]] but it requires C++14
|
||||
#define simdjson_deprecated __attribute__((deprecated))
|
||||
|
||||
#define simdjson_really_inline inline __attribute__((always_inline))
|
||||
#define simdjson_never_inline inline __attribute__((noinline))
|
||||
|
||||
@@ -13,6 +13,16 @@
|
||||
#endif
|
||||
#endif
|
||||
|
||||
// C++ 23
|
||||
#if !defined(SIMDJSON_CPLUSPLUS23) && (SIMDJSON_CPLUSPLUS >= 202302L)
|
||||
#define SIMDJSON_CPLUSPLUS23 1
|
||||
#endif
|
||||
|
||||
// C++ 20
|
||||
#if !defined(SIMDJSON_CPLUSPLUS20) && (SIMDJSON_CPLUSPLUS >= 202002L)
|
||||
#define SIMDJSON_CPLUSPLUS20 1
|
||||
#endif
|
||||
|
||||
// C++ 17
|
||||
#if !defined(SIMDJSON_CPLUSPLUS17) && (SIMDJSON_CPLUSPLUS >= 201703L)
|
||||
#define SIMDJSON_CPLUSPLUS17 1
|
||||
|
||||
@@ -123,6 +123,10 @@ inline simdjson_result<element> array::at(size_t index) const noexcept {
|
||||
return INDEX_OUT_OF_BOUNDS;
|
||||
}
|
||||
|
||||
inline array::operator element() const noexcept {
|
||||
return element(tape);
|
||||
}
|
||||
|
||||
//
|
||||
// array::iterator inline implementation
|
||||
//
|
||||
|
||||
@@ -126,6 +126,11 @@ public:
|
||||
*/
|
||||
inline simdjson_result<element> at(size_t index) const noexcept;
|
||||
|
||||
/**
|
||||
* Implicitly convert object to element
|
||||
*/
|
||||
inline operator element() const noexcept;
|
||||
|
||||
private:
|
||||
simdjson_inline array(const internal::tape_ref &tape) noexcept;
|
||||
internal::tape_ref tape;
|
||||
|
||||
@@ -153,6 +153,10 @@ inline simdjson_result<element> object::at_key_case_insensitive(std::string_view
|
||||
return NO_SUCH_FIELD;
|
||||
}
|
||||
|
||||
inline object::operator element() const noexcept {
|
||||
return element(tape);
|
||||
}
|
||||
|
||||
//
|
||||
// object::iterator inline implementation
|
||||
//
|
||||
|
||||
@@ -200,6 +200,11 @@ public:
|
||||
*/
|
||||
inline simdjson_result<element> at_key_case_insensitive(std::string_view key) const noexcept;
|
||||
|
||||
/**
|
||||
* Implicitly convert object to element
|
||||
*/
|
||||
inline operator element() const noexcept;
|
||||
|
||||
private:
|
||||
simdjson_inline object(const internal::tape_ref &tape) noexcept;
|
||||
|
||||
|
||||
@@ -3,16 +3,14 @@
|
||||
#ifndef SIMDJSON_CONDITIONAL_INCLUDE
|
||||
#define SIMDJSON_GENERIC_ONDEMAND_DOCUMENT_INL_H
|
||||
#include "simdjson/generic/ondemand/base.h"
|
||||
#include "simdjson/generic/ondemand/array-inl.h"
|
||||
#include "simdjson/generic/ondemand/array_iterator.h"
|
||||
#include "simdjson/generic/ondemand/document.h"
|
||||
#include "simdjson/generic/ondemand/json_iterator-inl.h"
|
||||
#include "simdjson/generic/ondemand/json_path_to_pointer_conversion.h"
|
||||
#include "simdjson/generic/ondemand/json_path_to_pointer_conversion-inl.h"
|
||||
#include "simdjson/generic/ondemand/json_type.h"
|
||||
#include "simdjson/generic/ondemand/object-inl.h"
|
||||
#include "simdjson/generic/ondemand/raw_json_string.h"
|
||||
#include "simdjson/generic/ondemand/value.h"
|
||||
#include "simdjson/generic/ondemand/array-inl.h"
|
||||
#include "simdjson/generic/ondemand/json_iterator-inl.h"
|
||||
#include "simdjson/generic/ondemand/object-inl.h"
|
||||
#include "simdjson/generic/ondemand/value_iterator-inl.h"
|
||||
#endif // SIMDJSON_CONDITIONAL_INCLUDE
|
||||
|
||||
@@ -167,24 +165,26 @@ template<> simdjson_inline simdjson_result<int64_t> document::get() & noexcept {
|
||||
template<> simdjson_inline simdjson_result<bool> document::get() & noexcept { return get_bool(); }
|
||||
template<> simdjson_inline simdjson_result<value> document::get() & noexcept { return get_value(); }
|
||||
|
||||
template<> simdjson_inline simdjson_result<raw_json_string> document::get() && noexcept { return get_raw_json_string(); }
|
||||
template<> simdjson_inline simdjson_result<std::string_view> document::get() && noexcept { return get_string(false); }
|
||||
template<> simdjson_inline simdjson_result<double> document::get() && noexcept { return std::forward<document>(*this).get_double(); }
|
||||
template<> simdjson_inline simdjson_result<uint64_t> document::get() && noexcept { return std::forward<document>(*this).get_uint64(); }
|
||||
template<> simdjson_inline simdjson_result<int64_t> document::get() && noexcept { return std::forward<document>(*this).get_int64(); }
|
||||
template<> simdjson_inline simdjson_result<bool> document::get() && noexcept { return std::forward<document>(*this).get_bool(); }
|
||||
template<> simdjson_inline simdjson_result<value> document::get() && noexcept { return get_value(); }
|
||||
template<> simdjson_deprecated simdjson_inline simdjson_result<raw_json_string> document::get() && noexcept { return get_raw_json_string(); }
|
||||
template<> simdjson_deprecated simdjson_inline simdjson_result<std::string_view> document::get() && noexcept { return get_string(false); }
|
||||
template<> simdjson_deprecated simdjson_inline simdjson_result<double> document::get() && noexcept { return std::forward<document>(*this).get_double(); }
|
||||
template<> simdjson_deprecated simdjson_inline simdjson_result<uint64_t> document::get() && noexcept { return std::forward<document>(*this).get_uint64(); }
|
||||
template<> simdjson_deprecated simdjson_inline simdjson_result<int64_t> document::get() && noexcept { return std::forward<document>(*this).get_int64(); }
|
||||
template<> simdjson_deprecated simdjson_inline simdjson_result<bool> document::get() && noexcept { return std::forward<document>(*this).get_bool(); }
|
||||
template<> simdjson_deprecated simdjson_inline simdjson_result<value> document::get() && noexcept { return get_value(); }
|
||||
|
||||
template<typename T> simdjson_inline error_code document::get(T &out) & noexcept {
|
||||
return get<T>().get(out);
|
||||
}
|
||||
template<typename T> simdjson_inline error_code document::get(T &out) && noexcept {
|
||||
template<typename T> simdjson_deprecated simdjson_inline error_code document::get(T &out) && noexcept {
|
||||
return std::forward<document>(*this).get<T>().get(out);
|
||||
}
|
||||
|
||||
#if SIMDJSON_EXCEPTIONS
|
||||
template <class T>
|
||||
simdjson_inline document::operator T() noexcept(false) { return get<T>(); }
|
||||
simdjson_deprecated simdjson_inline document::operator T() && noexcept(false) { return get<T>(); }
|
||||
template <class T>
|
||||
simdjson_inline document::operator T() & noexcept(false) { return get<T>(); }
|
||||
simdjson_inline document::operator array() & noexcept(false) { return get_array(); }
|
||||
simdjson_inline document::operator object() & noexcept(false) { return get_object(); }
|
||||
simdjson_inline document::operator uint64_t() noexcept(false) { return get_uint64(); }
|
||||
@@ -471,7 +471,7 @@ simdjson_inline simdjson_result<T> simdjson_result<SIMDJSON_IMPLEMENTATION::onde
|
||||
return first.get<T>();
|
||||
}
|
||||
template<typename T>
|
||||
simdjson_inline simdjson_result<T> simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::document>::get() && noexcept {
|
||||
simdjson_deprecated simdjson_inline simdjson_result<T> simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::document>::get() && noexcept {
|
||||
if (error()) { return error(); }
|
||||
return std::forward<SIMDJSON_IMPLEMENTATION::ondemand::document>(first).get<T>();
|
||||
}
|
||||
@@ -487,7 +487,7 @@ simdjson_inline error_code simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::do
|
||||
}
|
||||
|
||||
template<> simdjson_inline simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::document> simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::document>::get<SIMDJSON_IMPLEMENTATION::ondemand::document>() & noexcept = delete;
|
||||
template<> simdjson_inline simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::document> simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::document>::get<SIMDJSON_IMPLEMENTATION::ondemand::document>() && noexcept {
|
||||
template<> simdjson_deprecated simdjson_inline simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::document> simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::document>::get<SIMDJSON_IMPLEMENTATION::ondemand::document>() && noexcept {
|
||||
if (error()) { return error(); }
|
||||
return std::forward<SIMDJSON_IMPLEMENTATION::ondemand::document>(first);
|
||||
}
|
||||
|
||||
@@ -188,7 +188,7 @@ public:
|
||||
" You may also add support for custom types, see our documentation.");
|
||||
}
|
||||
/** @overload template<typename T> simdjson_result<T> get() & noexcept */
|
||||
template<typename T> simdjson_inline simdjson_result<T> get() && noexcept {
|
||||
template<typename T> simdjson_deprecated simdjson_inline simdjson_result<T> get() && noexcept {
|
||||
// Unless the simdjson library or the user provides an inline implementation, calling this method should
|
||||
// immediately fail.
|
||||
static_assert(!sizeof(T), "The get method with given type is not implemented by the simdjson library. "
|
||||
@@ -211,7 +211,7 @@ public:
|
||||
*/
|
||||
template<typename T> simdjson_inline error_code get(T &out) & noexcept;
|
||||
/** @overload template<typename T> error_code get(T &out) & noexcept */
|
||||
template<typename T> simdjson_inline error_code get(T &out) && noexcept;
|
||||
template<typename T> simdjson_deprecated simdjson_inline error_code get(T &out) && noexcept;
|
||||
|
||||
#if SIMDJSON_EXCEPTIONS
|
||||
/**
|
||||
@@ -224,7 +224,10 @@ public:
|
||||
* @returns An instance of type T
|
||||
*/
|
||||
template <class T>
|
||||
explicit simdjson_inline operator T() noexcept(false);
|
||||
explicit simdjson_inline operator T() & noexcept(false);
|
||||
template <class T>
|
||||
explicit simdjson_deprecated simdjson_inline operator T() && noexcept(false);
|
||||
|
||||
/**
|
||||
* Cast this JSON value to an array.
|
||||
*
|
||||
@@ -790,7 +793,7 @@ public:
|
||||
simdjson_inline simdjson_result<bool> is_null() noexcept;
|
||||
|
||||
template<typename T> simdjson_inline simdjson_result<T> get() & noexcept;
|
||||
template<typename T> simdjson_inline simdjson_result<T> get() && noexcept;
|
||||
template<typename T> simdjson_deprecated simdjson_inline simdjson_result<T> get() && noexcept;
|
||||
|
||||
template<typename T> simdjson_inline error_code get(T &out) & noexcept;
|
||||
template<typename T> simdjson_inline error_code get(T &out) && noexcept;
|
||||
|
||||
@@ -316,7 +316,7 @@ private:
|
||||
friend class document;
|
||||
friend class json_iterator;
|
||||
friend struct simdjson_result<ondemand::document_stream>;
|
||||
friend struct internal::simdjson_result_base<ondemand::document_stream>;
|
||||
friend struct simdjson::internal::simdjson_result_base<ondemand::document_stream>;
|
||||
}; // document_stream
|
||||
|
||||
} // namespace ondemand
|
||||
|
||||
@@ -38,6 +38,14 @@ simdjson_inline simdjson_warn_unused simdjson_result<std::string_view> field::un
|
||||
return answer;
|
||||
}
|
||||
|
||||
template <typename string_type>
|
||||
simdjson_inline simdjson_warn_unused error_code field::unescaped_key(string_type& receiver, bool allow_replacement) noexcept {
|
||||
std::string_view key;
|
||||
SIMDJSON_TRY( unescaped_key(allow_replacement).get(key) );
|
||||
receiver = key;
|
||||
return SUCCESS;
|
||||
}
|
||||
|
||||
simdjson_inline raw_json_string field::key() const noexcept {
|
||||
SIMDJSON_ASSUME(first.buf != nullptr); // We would like to call .alive() by Visual Studio won't let us.
|
||||
return first;
|
||||
@@ -105,6 +113,12 @@ simdjson_inline simdjson_result<std::string_view> simdjson_result<SIMDJSON_IMPLE
|
||||
return first.unescaped_key(allow_replacement);
|
||||
}
|
||||
|
||||
template<typename string_type>
|
||||
simdjson_inline error_code simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::field>::unescaped_key(string_type &receiver, bool allow_replacement) noexcept {
|
||||
if (error()) { return error(); }
|
||||
return first.unescaped_key(receiver, allow_replacement);
|
||||
}
|
||||
|
||||
simdjson_inline simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::value> simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::field>::value() noexcept {
|
||||
if (error()) { return error(); }
|
||||
return std::move(first.value());
|
||||
|
||||
@@ -36,21 +36,37 @@ public:
|
||||
* This consumes the key: once you have called unescaped_key(), you cannot
|
||||
* call it again nor can you call key().
|
||||
*/
|
||||
simdjson_inline simdjson_warn_unused simdjson_result<std::string_view> unescaped_key(bool allow_replacement) noexcept;
|
||||
simdjson_inline simdjson_warn_unused simdjson_result<std::string_view> unescaped_key(bool allow_replacement = false) noexcept;
|
||||
/**
|
||||
* Get the key as a string_view (for higher speed, consider raw_key).
|
||||
* We deliberately use a more cumbersome name (unescaped_key) to force users
|
||||
* to think twice about using it. The content is stored in the receiver.
|
||||
*
|
||||
* This consumes the key: once you have called unescaped_key(), you cannot
|
||||
* call it again nor can you call key().
|
||||
*/
|
||||
template <typename string_type>
|
||||
simdjson_inline simdjson_warn_unused error_code unescaped_key(string_type& receiver, bool allow_replacement = false) noexcept;
|
||||
/**
|
||||
* Get the key as a raw_json_string. Can be used for direct comparison with
|
||||
* an unescaped C string: e.g., key() == "test".
|
||||
* an unescaped C string: e.g., key() == "test". This does not count as
|
||||
* consumption of the content: you can safely call it repeatedly.
|
||||
* See escaped_key() for a similar function which returns
|
||||
* a more convenient std::string_view result.
|
||||
*/
|
||||
simdjson_inline raw_json_string key() const noexcept;
|
||||
/**
|
||||
* Get the unprocessed key as a string_view. This includes the quotes and may include
|
||||
* some spaces after the last quote.
|
||||
* some spaces after the last quote. This does not count as
|
||||
* consumption of the content: you can safely call it repeatedly.
|
||||
* See escaped_key().
|
||||
*/
|
||||
simdjson_inline std::string_view key_raw_json_token() const noexcept;
|
||||
/**
|
||||
* Get the key as a string_view. This does not include the quotes and
|
||||
* the string is unprocessed key so it may contain escape characters
|
||||
* (e.g., \uXXXX or \n). Use unescaped_key() to get the unescaped key.
|
||||
* (e.g., \uXXXX or \n). It does not count as a consumption of the content:
|
||||
* you can safely call it repeatedly. Use unescaped_key() to get the unescaped key.
|
||||
*/
|
||||
simdjson_inline std::string_view escaped_key() const noexcept;
|
||||
/**
|
||||
@@ -84,6 +100,8 @@ public:
|
||||
simdjson_inline simdjson_result() noexcept = default;
|
||||
|
||||
simdjson_inline simdjson_result<std::string_view> unescaped_key(bool allow_replacement = false) noexcept;
|
||||
template<typename string_type>
|
||||
simdjson_inline error_code unescaped_key(string_type &receiver, bool allow_replacement = false) noexcept;
|
||||
simdjson_inline simdjson_result<SIMDJSON_IMPLEMENTATION::ondemand::raw_json_string> key() noexcept;
|
||||
simdjson_inline simdjson_result<std::string_view> key_raw_json_token() noexcept;
|
||||
simdjson_inline simdjson_result<std::string_view> escaped_key() noexcept;
|
||||
|
||||
@@ -354,11 +354,23 @@ simdjson_inline token_position json_iterator::position() const noexcept {
|
||||
}
|
||||
|
||||
simdjson_inline simdjson_result<std::string_view> json_iterator::unescape(raw_json_string in, bool allow_replacement) noexcept {
|
||||
#if SIMDJSON_DEVELOPMENT_CHECKS
|
||||
auto result = parser->unescape(in, _string_buf_loc, allow_replacement);
|
||||
SIMDJSON_ASSUME(!parser->string_buffer_overflow(_string_buf_loc));
|
||||
return result;
|
||||
#else
|
||||
return parser->unescape(in, _string_buf_loc, allow_replacement);
|
||||
#endif
|
||||
}
|
||||
|
||||
simdjson_inline simdjson_result<std::string_view> json_iterator::unescape_wobbly(raw_json_string in) noexcept {
|
||||
#if SIMDJSON_DEVELOPMENT_CHECKS
|
||||
auto result = parser->unescape_wobbly(in, _string_buf_loc);
|
||||
SIMDJSON_ASSUME(!parser->string_buffer_overflow(_string_buf_loc));
|
||||
return result;
|
||||
#else
|
||||
return parser->unescape_wobbly(in, _string_buf_loc);
|
||||
#endif
|
||||
}
|
||||
|
||||
simdjson_inline void json_iterator::reenter_child(token_position position, depth_t child_depth) noexcept {
|
||||
|
||||
@@ -1,67 +0,0 @@
|
||||
#pragma once
|
||||
|
||||
#ifndef SIMDJSON_ONDEMAND_GENERIC_JSON_PATH_TO_POINTER_CONVERSION_INL_H
|
||||
#define SIMDJSON_ONDEMAND_GENERIC_JSON_PATH_TO_POINTER_CONVERSION_INL_H
|
||||
|
||||
#ifndef SIMDJSON_CONDITIONAL_INCLUDE
|
||||
#define SIMDJSON_GENERIC_ONDEMAND_JSON_PATH_TO_POINTER_CONVERSION_INL_H
|
||||
#include "simdjson/generic/ondemand/json_path_to_pointer_conversion.h"
|
||||
#endif // SIMDJSON_CONDITIONAL_INCLUDE
|
||||
|
||||
namespace simdjson {
|
||||
namespace SIMDJSON_IMPLEMENTATION {}
|
||||
namespace ondemand {
|
||||
|
||||
simdjson_inline std::string json_path_to_pointer_conversion(std::string_view json_path) {
|
||||
if (json_path.empty() || (json_path.front() != '.' && json_path.front() != '[') {
|
||||
return "-1"; // Sentinel value to be handled as an error by the caller.
|
||||
}
|
||||
|
||||
std::string result;
|
||||
// Reserve space to reduce allocations, adjusting for potential increases due
|
||||
// to escaping.
|
||||
result.reserve(json_path.size() * 2);
|
||||
|
||||
// Skip the initial '.' as it's assumed every path starts with it.
|
||||
size_t i = 0;
|
||||
|
||||
while (i < json_path.length()) {
|
||||
if (json_path[i] == '.') {
|
||||
result += '/';
|
||||
} else if (json_path[i] == '[') {
|
||||
result += '/';
|
||||
++i; // Move past the '['
|
||||
while (i < json_path.length() && json_path[i] != ']') {
|
||||
if (json_path[i] == '~') {
|
||||
result += "~0";
|
||||
} else if (json_path[i] == '/') {
|
||||
result += "~1";
|
||||
} else {
|
||||
result += json_path[i];
|
||||
}
|
||||
++i;
|
||||
}
|
||||
if (i == json_path.length() || json_path[i] != ']') {
|
||||
return "-1"; // Returning sentinel value that will be handled as an error by the caller
|
||||
}
|
||||
} else {
|
||||
if (json_path[i] == '~') {
|
||||
result += "~0";
|
||||
} else if (json_path[i] == '/') {
|
||||
result += "~1";
|
||||
} else {
|
||||
result += json_path[i];
|
||||
}
|
||||
}
|
||||
++i;
|
||||
}
|
||||
|
||||
return simdjson_result<std::string>(result);
|
||||
}
|
||||
|
||||
|
||||
} // namespace ondemand
|
||||
} // namespace SIMDJSON_IMPLEMENTATION
|
||||
} // namespace simdjson
|
||||
|
||||
#endif // SIMDJSON_ONDEMAND_GENERIC_JSON_PATH_TO_POINTER_CONVERSION_INL_H
|
||||
@@ -1,22 +0,0 @@
|
||||
#pragma once
|
||||
|
||||
#ifndef SIMDJSON_ONDEMAND_GENERIC_JSON_PATH_TO_POINTER_CONVERSION_H
|
||||
#define SIMDJSON_ONDEMAND_GENERIC_JSON_PATH_TO_POINTER_CONVERSION_H
|
||||
|
||||
namespace simdjson {
|
||||
namespace SIMDJSON_IMPLEMENTATION {
|
||||
namespace internal {
|
||||
|
||||
/**
|
||||
* Converts JSONPath to JSON Pointer.
|
||||
* @param json_path The JSONPath string to be converted.
|
||||
* @return A string containing the equivalent JSON Pointer.
|
||||
* @throws simdjson_error If the conversion fails.
|
||||
*/
|
||||
simdjson_inline std::string json_path_to_pointer_conversion(std::string_view json_path);
|
||||
|
||||
} // namespace internal
|
||||
} // namespace SIMDJSON_IMPLEMENTATION
|
||||
} // namespace simdjson
|
||||
|
||||
#endif // SIMDJSON_JSON_PATH_TO_POINTER_CONVERSION_H
|
||||
@@ -160,8 +160,8 @@ public:
|
||||
/**
|
||||
* Reset the iterator so that we are pointing back at the
|
||||
* beginning of the object. You should still consume values only once even if you
|
||||
* can iterate through the object more than once. If you unescape a string within
|
||||
* the object more than once, you have unsafe code. Note that rewinding an object
|
||||
* can iterate through the object more than once. If you unescape a string or a key
|
||||
* within the object more than once, you have unsafe code. Note that rewinding an object
|
||||
* means that you may need to reparse it anew: it is not a free operation.
|
||||
*
|
||||
* @returns true if the object contains some elements (not empty)
|
||||
|
||||
@@ -42,6 +42,11 @@ simdjson_warn_unused simdjson_inline error_code parser::allocate(size_t new_capa
|
||||
_max_depth = new_max_depth;
|
||||
return SUCCESS;
|
||||
}
|
||||
#if SIMDJSON_DEVELOPMENT_CHECKS
|
||||
simdjson_inline simdjson_warn_unused bool parser::string_buffer_overflow(const uint8_t *string_buf_loc) const noexcept {
|
||||
return (string_buf_loc < string_buf.get()) || (size_t(string_buf_loc - string_buf.get()) >= capacity());
|
||||
}
|
||||
#endif
|
||||
|
||||
simdjson_warn_unused simdjson_inline simdjson_result<document> parser::iterate(padded_string_view json) & noexcept {
|
||||
if (json.padding() < SIMDJSON_PADDING) { return INSUFFICIENT_PADDING; }
|
||||
|
||||
@@ -13,8 +13,8 @@ namespace SIMDJSON_IMPLEMENTATION {
|
||||
namespace ondemand {
|
||||
|
||||
/**
|
||||
* The default batch size for document_stream instances for this On Demand kernel.
|
||||
* Note that different On Demand kernel may use a different DEFAULT_BATCH_SIZE value
|
||||
* The default batch size for document_stream instances for this On-Demand kernel.
|
||||
* Note that different On-Demand kernel may use a different DEFAULT_BATCH_SIZE value
|
||||
* in the future.
|
||||
*/
|
||||
static constexpr size_t DEFAULT_BATCH_SIZE = 1000000;
|
||||
@@ -327,9 +327,20 @@ public:
|
||||
*/
|
||||
simdjson_inline simdjson_result<std::string_view> unescape_wobbly(raw_json_string in, uint8_t *&dst) const noexcept;
|
||||
|
||||
#if SIMDJSON_DEVELOPMENT_CHECKS
|
||||
/**
|
||||
* Returns true if string_buf_loc is outside of the allocated range for the
|
||||
* the string buffer. When true, it indicates that the string buffer has overflowed.
|
||||
* This is a development-time check that is not needed in production. It can be
|
||||
* used to detect buffer overflows in the string buffer and usafe usage of the
|
||||
* string buffer.
|
||||
*/
|
||||
bool string_buffer_overflow(const uint8_t *string_buf_loc) const noexcept;
|
||||
#endif
|
||||
|
||||
private:
|
||||
/** @private [for benchmarking access] The implementation to use */
|
||||
std::unique_ptr<internal::dom_parser_implementation> implementation{};
|
||||
std::unique_ptr<simdjson::internal::dom_parser_implementation> implementation{};
|
||||
size_t _capacity{0};
|
||||
size_t _max_capacity;
|
||||
size_t _max_depth{DEFAULT_MAX_DEPTH};
|
||||
|
||||
@@ -6,8 +6,6 @@
|
||||
#include "simdjson/generic/ondemand/array.h"
|
||||
#include "simdjson/generic/ondemand/array_iterator.h"
|
||||
#include "simdjson/generic/ondemand/json_iterator.h"
|
||||
#include "simdjson/generic/ondemand/json_path_to_pointer_conversion.h"
|
||||
#include "simdjson/generic/ondemand/json_path_to_pointer_conversion-inl.h"
|
||||
#include "simdjson/generic/ondemand/json_type.h"
|
||||
#include "simdjson/generic/ondemand/object.h"
|
||||
#include "simdjson/generic/ondemand/raw_json_string.h"
|
||||
|
||||
@@ -6,9 +6,9 @@
|
||||
#include "simdjson/generic/atomparsing.h"
|
||||
#include "simdjson/generic/numberparsing.h"
|
||||
#include "simdjson/generic/ondemand/json_iterator.h"
|
||||
#include "simdjson/generic/ondemand/value_iterator.h"
|
||||
#include "simdjson/generic/ondemand/json_type-inl.h"
|
||||
#include "simdjson/generic/ondemand/raw_json_string-inl.h"
|
||||
#include "simdjson/generic/ondemand/value_iterator.h"
|
||||
#endif // SIMDJSON_CONDITIONAL_INCLUDE
|
||||
|
||||
namespace simdjson {
|
||||
|
||||
@@ -4,7 +4,7 @@
|
||||
#include "simdjson/haswell/intrinsics.h"
|
||||
|
||||
#if !SIMDJSON_CAN_ALWAYS_RUN_HASWELL
|
||||
SIMDJSON_TARGET_REGION("avx2,bmi,pclmul,lzcnt,popcnt")
|
||||
SIMDJSON_TARGET_REGION("avx2,bmi,bmi2,pclmul,lzcnt,popcnt")
|
||||
#endif
|
||||
|
||||
#include "simdjson/haswell/bitmanipulation.h"
|
||||
|
||||
@@ -106,7 +106,7 @@ public:
|
||||
* Unescape a valid UTF-8 string from src to dst, stopping at a final unescaped quote. There
|
||||
* must be an unescaped quote terminating the string. It returns the final output
|
||||
* position as pointer. In case of error (e.g., the string has bad escaped codes),
|
||||
* then null_nullptrptr is returned. It is assumed that the output buffer is large
|
||||
* then null_ptr is returned. It is assumed that the output buffer is large
|
||||
* enough. E.g., if src points at 'joe"', then dst needs to have four free bytes +
|
||||
* SIMDJSON_PADDING bytes.
|
||||
*
|
||||
@@ -123,7 +123,7 @@ public:
|
||||
* Unescape a NON-valid UTF-8 string from src to dst, stopping at a final unescaped quote. There
|
||||
* must be an unescaped quote terminating the string. It returns the final output
|
||||
* position as pointer. In case of error (e.g., the string has bad escaped codes),
|
||||
* then null_nullptrptr is returned. It is assumed that the output buffer is large
|
||||
* then null_ptr is returned. It is assumed that the output buffer is large
|
||||
* enough. E.g., if src points at 'joe"', then dst needs to have four free bytes +
|
||||
* SIMDJSON_PADDING bytes.
|
||||
*
|
||||
|
||||
@@ -4,7 +4,7 @@
|
||||
#define SIMDJSON_SIMDJSON_VERSION_H
|
||||
|
||||
/** The version of simdjson being used (major.minor.revision) */
|
||||
#define SIMDJSON_VERSION "3.9.4"
|
||||
#define SIMDJSON_VERSION "3.10.1"
|
||||
|
||||
namespace simdjson {
|
||||
enum {
|
||||
@@ -15,11 +15,11 @@ enum {
|
||||
/**
|
||||
* The minor version (major.MINOR.revision) of simdjson being used.
|
||||
*/
|
||||
SIMDJSON_VERSION_MINOR = 9,
|
||||
SIMDJSON_VERSION_MINOR = 10,
|
||||
/**
|
||||
* The revision (major.minor.REVISION) of simdjson being used.
|
||||
*/
|
||||
SIMDJSON_VERSION_REVISION = 4
|
||||
SIMDJSON_VERSION_REVISION = 1
|
||||
};
|
||||
} // namespace simdjson
|
||||
|
||||
|
||||
+41
-22
@@ -1,4 +1,4 @@
|
||||
/* auto-generated on 2024-06-11 14:08:20 -0400. Do not edit! */
|
||||
/* auto-generated on 2024-08-26 09:37:03 -0400. Do not edit! */
|
||||
/* including simdjson.cpp: */
|
||||
/* begin file simdjson.cpp */
|
||||
#define SIMDJSON_SRC_SIMDJSON_CPP
|
||||
@@ -40,6 +40,16 @@
|
||||
#endif
|
||||
#endif
|
||||
|
||||
// C++ 23
|
||||
#if !defined(SIMDJSON_CPLUSPLUS23) && (SIMDJSON_CPLUSPLUS >= 202302L)
|
||||
#define SIMDJSON_CPLUSPLUS23 1
|
||||
#endif
|
||||
|
||||
// C++ 20
|
||||
#if !defined(SIMDJSON_CPLUSPLUS20) && (SIMDJSON_CPLUSPLUS >= 202002L)
|
||||
#define SIMDJSON_CPLUSPLUS20 1
|
||||
#endif
|
||||
|
||||
// C++ 17
|
||||
#if !defined(SIMDJSON_CPLUSPLUS17) && (SIMDJSON_CPLUSPLUS >= 201703L)
|
||||
#define SIMDJSON_CPLUSPLUS17 1
|
||||
@@ -224,6 +234,11 @@ using std::size_t;
|
||||
#define SIMDJSON_NO_SANITIZE_UNDEFINED
|
||||
#endif
|
||||
|
||||
#if defined(__clang__) || defined(__GNUC__)
|
||||
#define simdjson_pure [[gnu::pure]]
|
||||
#else
|
||||
#define simdjson_pure
|
||||
#endif
|
||||
|
||||
#if defined(__clang__) || defined(__GNUC__)
|
||||
#if defined(__has_feature)
|
||||
@@ -317,6 +332,8 @@ double from_chars(const char *first, const char* end) noexcept;
|
||||
#define SIMDJSON_ISALIGNED_N(ptr, n) (((uintptr_t)(ptr) & ((n)-1)) == 0)
|
||||
|
||||
#if SIMDJSON_REGULAR_VISUAL_STUDIO
|
||||
// We could use [[deprecated]] but it requires C++14
|
||||
#define simdjson_deprecated __declspec(deprecated)
|
||||
|
||||
#define simdjson_really_inline __forceinline
|
||||
#define simdjson_never_inline __declspec(noinline)
|
||||
@@ -355,6 +372,8 @@ double from_chars(const char *first, const char* end) noexcept;
|
||||
#define SIMDJSON_POP_DISABLE_UNUSED_WARNINGS
|
||||
|
||||
#else // SIMDJSON_REGULAR_VISUAL_STUDIO
|
||||
// We could use [[deprecated]] but it requires C++14
|
||||
#define simdjson_deprecated __attribute__((deprecated))
|
||||
|
||||
#define simdjson_really_inline inline __attribute__((always_inline))
|
||||
#define simdjson_never_inline inline __attribute__((noinline))
|
||||
@@ -5949,7 +5968,7 @@ public:
|
||||
* Unescape a valid UTF-8 string from src to dst, stopping at a final unescaped quote. There
|
||||
* must be an unescaped quote terminating the string. It returns the final output
|
||||
* position as pointer. In case of error (e.g., the string has bad escaped codes),
|
||||
* then null_nullptrptr is returned. It is assumed that the output buffer is large
|
||||
* then null_ptr is returned. It is assumed that the output buffer is large
|
||||
* enough. E.g., if src points at 'joe"', then dst needs to have four free bytes +
|
||||
* SIMDJSON_PADDING bytes.
|
||||
*
|
||||
@@ -5966,7 +5985,7 @@ public:
|
||||
* Unescape a NON-valid UTF-8 string from src to dst, stopping at a final unescaped quote. There
|
||||
* must be an unescaped quote terminating the string. It returns the final output
|
||||
* position as pointer. In case of error (e.g., the string has bad escaped codes),
|
||||
* then null_nullptrptr is returned. It is assumed that the output buffer is large
|
||||
* then null_ptr is returned. It is assumed that the output buffer is large
|
||||
* enough. E.g., if src points at 'joe"', then dst needs to have four free bytes +
|
||||
* SIMDJSON_PADDING bytes.
|
||||
*
|
||||
@@ -6020,14 +6039,14 @@ public:
|
||||
*
|
||||
* @return Current capacity, in bytes.
|
||||
*/
|
||||
simdjson_inline size_t capacity() const noexcept;
|
||||
simdjson_pure simdjson_inline size_t capacity() const noexcept;
|
||||
|
||||
/**
|
||||
* The maximum level of nested object and arrays supported by this parser.
|
||||
*
|
||||
* @return Maximum depth, in bytes.
|
||||
*/
|
||||
simdjson_inline size_t max_depth() const noexcept;
|
||||
simdjson_pure simdjson_inline size_t max_depth() const noexcept;
|
||||
|
||||
/**
|
||||
* Ensure this parser has enough memory to process JSON documents up to `capacity` bytes in length
|
||||
@@ -6068,11 +6087,11 @@ simdjson_inline dom_parser_implementation::dom_parser_implementation() noexcept
|
||||
simdjson_inline dom_parser_implementation::dom_parser_implementation(dom_parser_implementation &&other) noexcept = default;
|
||||
simdjson_inline dom_parser_implementation &dom_parser_implementation::operator=(dom_parser_implementation &&other) noexcept = default;
|
||||
|
||||
simdjson_inline size_t dom_parser_implementation::capacity() const noexcept {
|
||||
simdjson_pure simdjson_inline size_t dom_parser_implementation::capacity() const noexcept {
|
||||
return _capacity;
|
||||
}
|
||||
|
||||
simdjson_inline size_t dom_parser_implementation::max_depth() const noexcept {
|
||||
simdjson_pure simdjson_inline size_t dom_parser_implementation::max_depth() const noexcept {
|
||||
return _max_depth;
|
||||
}
|
||||
|
||||
@@ -12477,7 +12496,7 @@ simdjson_inline error_code json_structural_indexer::finish(dom_parser_implementa
|
||||
}
|
||||
parser.n_structural_indexes = uint32_t(indexer.tail - parser.structural_indexes.get());
|
||||
/***
|
||||
* The On Demand API requires special padding.
|
||||
* The On-Demand API requires special padding.
|
||||
*
|
||||
* This is related to https://github.com/simdjson/simdjson/issues/906
|
||||
* Basically, we want to make sure that if the parsing continues beyond the last (valid)
|
||||
@@ -13352,7 +13371,7 @@ simdjson_inline bool handle_unicode_codepoint_wobbly(const uint8_t **src_ptr,
|
||||
* Unescape a valid UTF-8 string from src to dst, stopping at a final unescaped quote. There
|
||||
* must be an unescaped quote terminating the string. It returns the final output
|
||||
* position as pointer. In case of error (e.g., the string has bad escaped codes),
|
||||
* then null_nullptrptr is returned. It is assumed that the output buffer is large
|
||||
* then null_ptr is returned. It is assumed that the output buffer is large
|
||||
* enough. E.g., if src points at 'joe"', then dst needs to have four free bytes +
|
||||
* SIMDJSON_PADDING bytes.
|
||||
*/
|
||||
@@ -18696,7 +18715,7 @@ simdjson_inline error_code json_structural_indexer::finish(dom_parser_implementa
|
||||
}
|
||||
parser.n_structural_indexes = uint32_t(indexer.tail - parser.structural_indexes.get());
|
||||
/***
|
||||
* The On Demand API requires special padding.
|
||||
* The On-Demand API requires special padding.
|
||||
*
|
||||
* This is related to https://github.com/simdjson/simdjson/issues/906
|
||||
* Basically, we want to make sure that if the parsing continues beyond the last (valid)
|
||||
@@ -19571,7 +19590,7 @@ simdjson_inline bool handle_unicode_codepoint_wobbly(const uint8_t **src_ptr,
|
||||
* Unescape a valid UTF-8 string from src to dst, stopping at a final unescaped quote. There
|
||||
* must be an unescaped quote terminating the string. It returns the final output
|
||||
* position as pointer. In case of error (e.g., the string has bad escaped codes),
|
||||
* then null_nullptrptr is returned. It is assumed that the output buffer is large
|
||||
* then null_ptr is returned. It is assumed that the output buffer is large
|
||||
* enough. E.g., if src points at 'joe"', then dst needs to have four free bytes +
|
||||
* SIMDJSON_PADDING bytes.
|
||||
*/
|
||||
@@ -24908,7 +24927,7 @@ simdjson_inline error_code json_structural_indexer::finish(dom_parser_implementa
|
||||
}
|
||||
parser.n_structural_indexes = uint32_t(indexer.tail - parser.structural_indexes.get());
|
||||
/***
|
||||
* The On Demand API requires special padding.
|
||||
* The On-Demand API requires special padding.
|
||||
*
|
||||
* This is related to https://github.com/simdjson/simdjson/issues/906
|
||||
* Basically, we want to make sure that if the parsing continues beyond the last (valid)
|
||||
@@ -25783,7 +25802,7 @@ simdjson_inline bool handle_unicode_codepoint_wobbly(const uint8_t **src_ptr,
|
||||
* Unescape a valid UTF-8 string from src to dst, stopping at a final unescaped quote. There
|
||||
* must be an unescaped quote terminating the string. It returns the final output
|
||||
* position as pointer. In case of error (e.g., the string has bad escaped codes),
|
||||
* then null_nullptrptr is returned. It is assumed that the output buffer is large
|
||||
* then null_ptr is returned. It is assumed that the output buffer is large
|
||||
* enough. E.g., if src points at 'joe"', then dst needs to have four free bytes +
|
||||
* SIMDJSON_PADDING bytes.
|
||||
*/
|
||||
@@ -31391,7 +31410,7 @@ simdjson_inline error_code json_structural_indexer::finish(dom_parser_implementa
|
||||
}
|
||||
parser.n_structural_indexes = uint32_t(indexer.tail - parser.structural_indexes.get());
|
||||
/***
|
||||
* The On Demand API requires special padding.
|
||||
* The On-Demand API requires special padding.
|
||||
*
|
||||
* This is related to https://github.com/simdjson/simdjson/issues/906
|
||||
* Basically, we want to make sure that if the parsing continues beyond the last (valid)
|
||||
@@ -32266,7 +32285,7 @@ simdjson_inline bool handle_unicode_codepoint_wobbly(const uint8_t **src_ptr,
|
||||
* Unescape a valid UTF-8 string from src to dst, stopping at a final unescaped quote. There
|
||||
* must be an unescaped quote terminating the string. It returns the final output
|
||||
* position as pointer. In case of error (e.g., the string has bad escaped codes),
|
||||
* then null_nullptrptr is returned. It is assumed that the output buffer is large
|
||||
* then null_ptr is returned. It is assumed that the output buffer is large
|
||||
* enough. E.g., if src points at 'joe"', then dst needs to have four free bytes +
|
||||
* SIMDJSON_PADDING bytes.
|
||||
*/
|
||||
@@ -38448,7 +38467,7 @@ simdjson_inline error_code json_structural_indexer::finish(dom_parser_implementa
|
||||
}
|
||||
parser.n_structural_indexes = uint32_t(indexer.tail - parser.structural_indexes.get());
|
||||
/***
|
||||
* The On Demand API requires special padding.
|
||||
* The On-Demand API requires special padding.
|
||||
*
|
||||
* This is related to https://github.com/simdjson/simdjson/issues/906
|
||||
* Basically, we want to make sure that if the parsing continues beyond the last (valid)
|
||||
@@ -39323,7 +39342,7 @@ simdjson_inline bool handle_unicode_codepoint_wobbly(const uint8_t **src_ptr,
|
||||
* Unescape a valid UTF-8 string from src to dst, stopping at a final unescaped quote. There
|
||||
* must be an unescaped quote terminating the string. It returns the final output
|
||||
* position as pointer. In case of error (e.g., the string has bad escaped codes),
|
||||
* then null_nullptrptr is returned. It is assumed that the output buffer is large
|
||||
* then null_ptr is returned. It is assumed that the output buffer is large
|
||||
* enough. E.g., if src points at 'joe"', then dst needs to have four free bytes +
|
||||
* SIMDJSON_PADDING bytes.
|
||||
*/
|
||||
@@ -44472,7 +44491,7 @@ simdjson_inline error_code json_structural_indexer::finish(dom_parser_implementa
|
||||
}
|
||||
parser.n_structural_indexes = uint32_t(indexer.tail - parser.structural_indexes.get());
|
||||
/***
|
||||
* The On Demand API requires special padding.
|
||||
* The On-Demand API requires special padding.
|
||||
*
|
||||
* This is related to https://github.com/simdjson/simdjson/issues/906
|
||||
* Basically, we want to make sure that if the parsing continues beyond the last (valid)
|
||||
@@ -45347,7 +45366,7 @@ simdjson_inline bool handle_unicode_codepoint_wobbly(const uint8_t **src_ptr,
|
||||
* Unescape a valid UTF-8 string from src to dst, stopping at a final unescaped quote. There
|
||||
* must be an unescaped quote terminating the string. It returns the final output
|
||||
* position as pointer. In case of error (e.g., the string has bad escaped codes),
|
||||
* then null_nullptrptr is returned. It is assumed that the output buffer is large
|
||||
* then null_ptr is returned. It is assumed that the output buffer is large
|
||||
* enough. E.g., if src points at 'joe"', then dst needs to have four free bytes +
|
||||
* SIMDJSON_PADDING bytes.
|
||||
*/
|
||||
@@ -50487,7 +50506,7 @@ simdjson_inline error_code json_structural_indexer::finish(dom_parser_implementa
|
||||
}
|
||||
parser.n_structural_indexes = uint32_t(indexer.tail - parser.structural_indexes.get());
|
||||
/***
|
||||
* The On Demand API requires special padding.
|
||||
* The On-Demand API requires special padding.
|
||||
*
|
||||
* This is related to https://github.com/simdjson/simdjson/issues/906
|
||||
* Basically, we want to make sure that if the parsing continues beyond the last (valid)
|
||||
@@ -51362,7 +51381,7 @@ simdjson_inline bool handle_unicode_codepoint_wobbly(const uint8_t **src_ptr,
|
||||
* Unescape a valid UTF-8 string from src to dst, stopping at a final unescaped quote. There
|
||||
* must be an unescaped quote terminating the string. It returns the final output
|
||||
* position as pointer. In case of error (e.g., the string has bad escaped codes),
|
||||
* then null_nullptrptr is returned. It is assumed that the output buffer is large
|
||||
* then null_ptr is returned. It is assumed that the output buffer is large
|
||||
* enough. E.g., if src points at 'joe"', then dst needs to have four free bytes +
|
||||
* SIMDJSON_PADDING bytes.
|
||||
*/
|
||||
@@ -54563,7 +54582,7 @@ simdjson_inline bool handle_unicode_codepoint_wobbly(const uint8_t **src_ptr,
|
||||
* Unescape a valid UTF-8 string from src to dst, stopping at a final unescaped quote. There
|
||||
* must be an unescaped quote terminating the string. It returns the final output
|
||||
* position as pointer. In case of error (e.g., the string has bad escaped codes),
|
||||
* then null_nullptrptr is returned. It is assumed that the output buffer is large
|
||||
* then null_ptr is returned. It is assumed that the output buffer is large
|
||||
* enough. E.g., if src points at 'joe"', then dst needs to have four free bytes +
|
||||
* SIMDJSON_PADDING bytes.
|
||||
*/
|
||||
|
||||
+850
-324
File diff suppressed because it is too large
Load Diff
@@ -263,7 +263,7 @@ simdjson_inline error_code json_structural_indexer::finish(dom_parser_implementa
|
||||
}
|
||||
parser.n_structural_indexes = uint32_t(indexer.tail - parser.structural_indexes.get());
|
||||
/***
|
||||
* The On Demand API requires special padding.
|
||||
* The On-Demand API requires special padding.
|
||||
*
|
||||
* This is related to https://github.com/simdjson/simdjson/issues/906
|
||||
* Basically, we want to make sure that if the parsing continues beyond the last (valid)
|
||||
|
||||
@@ -143,7 +143,7 @@ simdjson_inline bool handle_unicode_codepoint_wobbly(const uint8_t **src_ptr,
|
||||
* Unescape a valid UTF-8 string from src to dst, stopping at a final unescaped quote. There
|
||||
* must be an unescaped quote terminating the string. It returns the final output
|
||||
* position as pointer. In case of error (e.g., the string has bad escaped codes),
|
||||
* then null_nullptrptr is returned. It is assumed that the output buffer is large
|
||||
* then null_ptr is returned. It is assumed that the output buffer is large
|
||||
* enough. E.g., if src points at 'joe"', then dst needs to have four free bytes +
|
||||
* SIMDJSON_PADDING bytes.
|
||||
*/
|
||||
|
||||
@@ -38,6 +38,25 @@ const size_t AMAZON_CELLPHONES_NDJSON_DOC_COUNT = 793;
|
||||
|
||||
namespace number_tests {
|
||||
|
||||
bool build(const std::string& json) {
|
||||
simdjson::dom::parser parser;
|
||||
simdjson::dom::element recvdJson;
|
||||
auto error = parser.parse(json.c_str(), json.size()).get(recvdJson);
|
||||
if (error) {
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
bool issue2213() {
|
||||
TEST_START();
|
||||
std::string jsonStr = "[1,2,3,\"4\", {\"a\": 5}]";
|
||||
for (int i = 0; i < 15; ++i) {
|
||||
if (!build(jsonStr)) {
|
||||
TEST_FAIL("The JSON is valid");
|
||||
}
|
||||
}
|
||||
TEST_SUCCEED();
|
||||
}
|
||||
bool ground_truth() {
|
||||
std::cout << __func__ << std::endl;
|
||||
std::pair<std::string,double> ground_truth[] = {
|
||||
@@ -397,7 +416,8 @@ namespace number_tests {
|
||||
}
|
||||
|
||||
bool run() {
|
||||
return bomskip() &&
|
||||
return issue2213() &&
|
||||
bomskip() &&
|
||||
issue2017() &&
|
||||
truncated_borderline() &&
|
||||
specific_tests() &&
|
||||
@@ -950,6 +970,36 @@ namespace dom_api_tests {
|
||||
return true;
|
||||
}
|
||||
|
||||
bool convert_object_to_element() {
|
||||
std::cout << "Running " << __func__ << std::endl;
|
||||
string json(R"({ "a": 1, "b": 2, "c": 3 })");
|
||||
|
||||
dom::parser parser;
|
||||
dom::object object;
|
||||
dom::element element;
|
||||
ASSERT_SUCCESS( parser.parse(json).get(object) );
|
||||
element = object;
|
||||
ASSERT_EQUAL( element["a"].get_uint64().value_unsafe(), 1 );
|
||||
ASSERT_EQUAL( element["b"].get_uint64().value_unsafe(), 2 );
|
||||
ASSERT_EQUAL( element["c"].get_uint64().value_unsafe(), 3 );
|
||||
return true;
|
||||
}
|
||||
|
||||
bool convert_array_to_element() {
|
||||
std::cout << "Running " << __func__ << std::endl;
|
||||
string json(R"([ 1, 10, 100 ])");
|
||||
|
||||
dom::parser parser;
|
||||
dom::array array;
|
||||
dom::element element;
|
||||
ASSERT_SUCCESS( parser.parse(json).get(array) );
|
||||
element = array;
|
||||
ASSERT_EQUAL( element.at(0).get_uint64().value_unsafe(), 1 );
|
||||
ASSERT_EQUAL( element.at(1).get_uint64().value_unsafe(), 10 );
|
||||
ASSERT_EQUAL( element.at(2).get_uint64().value_unsafe(), 100 );
|
||||
return true;
|
||||
}
|
||||
|
||||
bool string_value() {
|
||||
std::cout << "Running " << __func__ << std::endl;
|
||||
string json(R"([ "hi", "has backslash\\" ])");
|
||||
@@ -1331,6 +1381,8 @@ namespace dom_api_tests {
|
||||
array_iterator_empty() &&
|
||||
object_iterator_advance() &&
|
||||
array_iterator_advance() &&
|
||||
convert_object_to_element() &&
|
||||
convert_array_to_element() &&
|
||||
string_value() &&
|
||||
numeric_values() &&
|
||||
boolean_values() &&
|
||||
|
||||
@@ -13,7 +13,8 @@ function(add_dual_compile_test TEST_NAME)
|
||||
target_compile_definitions(${TEST_NAME}_should_not_compile PRIVATE COMPILATION_TEST_USE_FAILING_CODE=1)
|
||||
endfunction(add_dual_compile_test)
|
||||
|
||||
|
||||
add_dual_compile_test(iterate_object)
|
||||
add_dual_compile_test(iterate_array)
|
||||
add_dual_compile_test(iterate_char_star)
|
||||
add_dual_compile_test(iterate_string_view)
|
||||
add_dual_compile_test(iterate_temporary_buffer)
|
||||
|
||||
@@ -0,0 +1,34 @@
|
||||
|
||||
#include <iostream>
|
||||
#include "simdjson.h"
|
||||
|
||||
using namespace simdjson;
|
||||
|
||||
int main() {
|
||||
auto json = "[1]"_padded;
|
||||
ondemand::parser parser;
|
||||
auto f = [](ondemand::parser& p, simdjson::padded_string& jsons) -> ondemand::document {
|
||||
ondemand::document doc;
|
||||
auto error = p.iterate(jsons).get(doc);
|
||||
if(error) { std::abort(); }
|
||||
return doc;
|
||||
};
|
||||
ondemand::array arrayv;
|
||||
#if COMPILATION_TEST_USE_FAILING_CODE
|
||||
// Not allowed as this would be unsafe, the document must remain alive.
|
||||
auto error = f(parser).get_array().get(arrayv);
|
||||
#else
|
||||
ondemand::document doc = f(parser, json);
|
||||
auto error = doc.get_array().get(arrayv);
|
||||
#endif
|
||||
if(error) {
|
||||
std::cout << "Failure" << std::endl;
|
||||
}
|
||||
int64_t a = 0;
|
||||
error = arrayv.at(0).get_int64().get(a);
|
||||
if(error) {
|
||||
std::cout << "failure" << std::endl;
|
||||
}
|
||||
printf("a = %d\n", (int)a);
|
||||
return EXIT_SUCCESS;
|
||||
}
|
||||
@@ -0,0 +1,34 @@
|
||||
|
||||
#include <iostream>
|
||||
#include "simdjson.h"
|
||||
|
||||
using namespace simdjson;
|
||||
|
||||
int main() {
|
||||
auto json = "{\"a\":1}"_padded;
|
||||
ondemand::parser parser;
|
||||
auto f = [](ondemand::parser& p, simdjson::padded_string& jsons) -> ondemand::document {
|
||||
ondemand::document doc;
|
||||
auto error = p.iterate(jsons).get(doc);
|
||||
if(error) { std::abort(); }
|
||||
return doc;
|
||||
};
|
||||
ondemand::object objv;
|
||||
#if COMPILATION_TEST_USE_FAILING_CODE
|
||||
// Not allowed as this would be unsafe, the document must remain alive.
|
||||
auto error = f(parser).get_object().get(objv);
|
||||
#else
|
||||
ondemand::document doc = f(parser, json);
|
||||
auto error = doc.get_object().get(objv);
|
||||
#endif
|
||||
if(error) {
|
||||
std::cout << "Failure" << std::endl;
|
||||
}
|
||||
int64_t a = 0;
|
||||
error = objv["a"].get_int64().get(a);
|
||||
if(error) {
|
||||
std::cout << "failure" << std::endl;
|
||||
}
|
||||
printf("a = %d\n", (int)a);
|
||||
return EXIT_SUCCESS;
|
||||
}
|
||||
@@ -59,6 +59,22 @@ void compilation_test_3() {
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Do not run this, it is only meant to compile
|
||||
void compilation_test_4() {
|
||||
const padded_string bogus = ""_padded;
|
||||
ondemand::parser parser;
|
||||
int64_t x1 = (int64_t)parser.iterate(bogus);
|
||||
double x2 = (double)parser.iterate(bogus);
|
||||
std::string_view x = (std::string_view)parser.iterate(bogus);
|
||||
uint64_t x3 = (uint64_t)parser.iterate(bogus);
|
||||
bool x4 = (bool)parser.iterate(bogus);
|
||||
(void) x1;
|
||||
(void) x2;
|
||||
(void) x3;
|
||||
(void) x4;
|
||||
(void) x;
|
||||
}
|
||||
#endif // SIMDJSON_EXCEPTIONS
|
||||
|
||||
int main(void) {
|
||||
|
||||
@@ -164,8 +164,7 @@ namespace json_pointer_tests {
|
||||
ASSERT_TRUE(is_scalar);
|
||||
ASSERT_ERROR(doc.at_pointer("").get(val), simdjson::SCALAR_DOCUMENT_AS_VALUE);
|
||||
std::cout << " checking true"<< std::endl;
|
||||
ASSERT_SUCCESS(parser.iterate(true_json).get(doc));
|
||||
ASSERT_SUCCESS(doc.is_scalar().get(is_scalar));
|
||||
ASSERT_SUCCESS(parser.iterate(true_json).is_scalar().get(is_scalar));
|
||||
ASSERT_TRUE(is_scalar);
|
||||
ASSERT_ERROR(doc.at_pointer("").get(val), simdjson::SCALAR_DOCUMENT_AS_VALUE);
|
||||
std::cout << " checking object"<< std::endl;
|
||||
|
||||
@@ -700,6 +700,22 @@ namespace object_tests {
|
||||
}
|
||||
return got_key;
|
||||
}));
|
||||
SUBTEST("ondemand::unescapedkey(std)", test_ondemand_doc(json, [&](auto doc_result) {
|
||||
ondemand::object object;
|
||||
bool got_key = false;
|
||||
ASSERT_SUCCESS( doc_result.get(object) );
|
||||
for (auto field : object) {
|
||||
std::string keyv;
|
||||
ASSERT_SUCCESS( field.unescaped_key(keyv) );
|
||||
if(keyv == "key") {
|
||||
int64_t value;
|
||||
ASSERT_SUCCESS( field.value().get(value) );
|
||||
ASSERT_EQUAL( value, 1);
|
||||
got_key = true;
|
||||
}
|
||||
}
|
||||
return got_key;
|
||||
}));
|
||||
SUBTEST("ondemand::rawkey", test_ondemand_doc(json, [&](auto doc_result) {
|
||||
ondemand::object object;
|
||||
ASSERT_SUCCESS( doc_result.get(object) );
|
||||
|
||||
@@ -1557,6 +1557,29 @@ bool allow_comma_separated_example() {
|
||||
}
|
||||
TEST_SUCCEED();
|
||||
}
|
||||
|
||||
bool issue2215() {
|
||||
TEST_START();
|
||||
ondemand::parser parser;
|
||||
const padded_string json = R"({ "parent": {"child1": {"name": "John"} , "child2": {"name": "Daniel"}} })"_padded;
|
||||
auto doc = parser.iterate(json);
|
||||
ondemand::object parent = doc["parent"];
|
||||
// parent owns the focus
|
||||
ondemand::object c1 = parent["child1"];
|
||||
// c1 owns the focus
|
||||
//
|
||||
std::string_view as1 = c1["name"];
|
||||
// We have that as1 == "John", as long as 'parser' and 'json' live
|
||||
// c2 attempts to grab the focus from parent but fails
|
||||
ondemand::object c2 = parent["child2"];
|
||||
// c2 owns the focus, at this point c1 is invalid
|
||||
std::string_view as2 = c2["name"];
|
||||
// We have that as2 == "Daniel", as long as 'parser' and 'json' live
|
||||
ASSERT_EQUAL(as1, "John");
|
||||
ASSERT_EQUAL(as2, "Daniel");
|
||||
std::cout << as1 << " " << as2 << std::endl;
|
||||
TEST_SUCCEED();
|
||||
}
|
||||
#endif
|
||||
bool test_load_example() {
|
||||
TEST_START();
|
||||
@@ -1918,6 +1941,7 @@ bool run() {
|
||||
&& current_location_no_error()
|
||||
&& to_string_example_no_except()
|
||||
#if SIMDJSON_EXCEPTIONS
|
||||
&& issue2215()
|
||||
&& to_string_example()
|
||||
&& raw_string()
|
||||
&& number_tests()
|
||||
|
||||
+2
-2
@@ -31,7 +31,7 @@ int main(int argc, const char *argv[]) {
|
||||
cxxopts::Options options(progName, progUsage);
|
||||
|
||||
options.add_options()
|
||||
("z,ondemand", "Use On Demand front-end.", cxxopts::value<bool>()->default_value("false"))
|
||||
("z,ondemand", "Use On-Demand front-end.", cxxopts::value<bool>()->default_value("false"))
|
||||
("d,rawdump", "Dumps the raw content of the tape.", cxxopts::value<bool>()->default_value("false"))
|
||||
("f,file", "File name.", cxxopts::value<std::string>())
|
||||
("h,help", "Print usage.")
|
||||
@@ -92,7 +92,7 @@ int main(int argc, const char *argv[]) {
|
||||
}
|
||||
return EXIT_SUCCESS;
|
||||
#ifdef __cpp_exceptions
|
||||
} catch (const cxxopts::OptionException& e) {
|
||||
} catch (const cxxopts::exceptions::option_has_no_value& e) {
|
||||
std::cout << "error parsing options: " << e.what() << std::endl;
|
||||
return EXIT_FAILURE;
|
||||
}
|
||||
|
||||
+1
-1
@@ -278,7 +278,7 @@ int main(int argc, const char *argv[]) {
|
||||
s.repeated_key_byte_count, s.maximum_depth);
|
||||
return EXIT_SUCCESS;
|
||||
#ifdef __cpp_exceptions
|
||||
} catch (const cxxopts::OptionException& e) {
|
||||
} catch (const cxxopts::exceptions::option_has_no_value& e) {
|
||||
std::cout << "error parsing options: " << e.what() << std::endl;
|
||||
return EXIT_FAILURE;
|
||||
}
|
||||
|
||||
+1
-1
@@ -111,7 +111,7 @@ int main(int argc, const char *argv[]) {
|
||||
}
|
||||
return EXIT_SUCCESS;
|
||||
#ifdef __cpp_exceptions
|
||||
} catch (const cxxopts::OptionException& e) {
|
||||
} catch (const cxxopts::exceptions::option_has_no_value& e) {
|
||||
std::cout << "error parsing options: " << e.what() << std::endl;
|
||||
return EXIT_FAILURE;
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user