mirror of
https://github.com/simdjson/simdjson
synced 2026-06-08 17:27:07 +00:00
126 lines
5.1 KiB
C++
126 lines
5.1 KiB
C++
// TODO Remove this -- deprecated API and files
|
|
|
|
#ifndef SIMDJSON_JSONSTREAM_H
|
|
#define SIMDJSON_JSONSTREAM_H
|
|
|
|
#include "simdjson/document_stream.h"
|
|
#include "simdjson/padded_string.h"
|
|
|
|
namespace simdjson {
|
|
|
|
/**
|
|
* @deprecated use document::stream instead.
|
|
*
|
|
* The main motivation for this piece of software is to achieve maximum speed and offer
|
|
* good quality of life while parsing files containing multiple JSON documents.
|
|
*
|
|
* Since we want to offer flexibility and not restrict ourselves to a specific file
|
|
* format, we support any file that contains any valid JSON documents separated by one
|
|
* or more character that is considered a whitespace by the JSON spec.
|
|
* Namely: space, nothing, linefeed, carriage return, horizontal tab.
|
|
* Anything that is not whitespace will be parsed as a JSON document and could lead
|
|
* to failure.
|
|
*
|
|
* To offer maximum parsing speed, our implementation processes the data inside the
|
|
* buffer by batches and their size is defined by the parameter "batch_size".
|
|
* By loading data in batches, we can optimize the time spent allocating data in the
|
|
* parser and can also open the possibility of multi-threading.
|
|
* The batch_size must be at least as large as the biggest document in the file, but
|
|
* not too large in order to submerge the chached memory. We found that 1MB is
|
|
* somewhat a sweet spot for now. Eventually, this batch_size could be fully
|
|
* automated and be optimal at all times.
|
|
*
|
|
* The template parameter (string_container) must
|
|
* support the data() and size() methods, returning a pointer
|
|
* to a char* and to the number of bytes respectively.
|
|
* The simdjson parser may read up to SIMDJSON_PADDING bytes beyond the end
|
|
* of the string, so if you do not use a padded_string container,
|
|
* you have the responsability to overallocated. If you fail to
|
|
* do so, your software may crash if you cross a page boundary,
|
|
* and you should expect memory checkers to object.
|
|
* Most users should use a simdjson::padded_string.
|
|
*/
|
|
template <class string_container> class JsonStream {
|
|
public:
|
|
/* Create a JsonStream object that can be used to parse sequentially the valid
|
|
* JSON documents found in the buffer "buf".
|
|
*
|
|
* The batch_size must be at least as large as the biggest document in the
|
|
* file, but
|
|
* not too large to submerge the cached memory. We found that 1MB is
|
|
* somewhat a sweet spot for now.
|
|
*
|
|
* The user is expected to call the following json_parse method to parse the
|
|
* next
|
|
* valid JSON document found in the buffer. This method can and is expected
|
|
* to be
|
|
* called in a loop.
|
|
*
|
|
* Various methods are offered to keep track of the status, like
|
|
* get_current_buffer_loc,
|
|
* get_n_parsed_docs, get_n_bytes_parsed, etc.
|
|
*
|
|
* */
|
|
JsonStream(const string_container &s, size_t _batch_size = 1000000) noexcept;
|
|
|
|
~JsonStream() noexcept;
|
|
|
|
/* Parse the next document found in the buffer previously given to JsonStream.
|
|
|
|
* The content should be a valid JSON document encoded as UTF-8. If there is a
|
|
* UTF-8 BOM, the caller is responsible for omitting it, UTF-8 BOM are
|
|
* discouraged.
|
|
*
|
|
* You do NOT need to pre-allocate a parser. This function takes care of
|
|
* pre-allocating a capacity defined by the batch_size defined when creating
|
|
the
|
|
* JsonStream object.
|
|
*
|
|
* The function returns simdjson::SUCCESS_AND_HAS_MORE (an integer = 1) in
|
|
case
|
|
* of success and indicates that the buffer still contains more data to be
|
|
parsed,
|
|
* meaning this function can be called again to return the next JSON document
|
|
* after this one.
|
|
*
|
|
* The function returns simdjson::SUCCESS (as integer = 0) in case of success
|
|
* and indicates that the buffer has successfully been parsed to the end.
|
|
* Every document it contained has been parsed without error.
|
|
*
|
|
* The function returns an error code from simdjson/simdjson.h in case of
|
|
failure
|
|
* such as simdjson::CAPACITY, simdjson::MEMALLOC, simdjson::DEPTH_ERROR and
|
|
so forth;
|
|
* the simdjson::error_message function converts these error codes into a
|
|
* string).
|
|
*
|
|
* You can also check validity by calling parser.is_valid(). The same parser
|
|
can
|
|
* and should be reused for the other documents in the buffer. */
|
|
int json_parse(document::parser &parser) noexcept;
|
|
|
|
/* Returns the location (index) of where the next document should be in the
|
|
* buffer.
|
|
* Can be used for debugging, it tells the user the position of the end of the
|
|
* last
|
|
* valid JSON document parsed*/
|
|
inline size_t get_current_buffer_loc() const noexcept { return stream ? stream->current_buffer_loc : 0; }
|
|
|
|
/* Returns the total amount of complete documents parsed by the JsonStream,
|
|
* in the current buffer, at the given time.*/
|
|
inline size_t get_n_parsed_docs() const noexcept { return stream ? stream->n_parsed_docs : 0; }
|
|
|
|
/* Returns the total amount of data (in bytes) parsed by the JsonStream,
|
|
* in the current buffer, at the given time.*/
|
|
inline size_t get_n_bytes_parsed() const noexcept { return stream ? stream->n_bytes_parsed : 0; }
|
|
|
|
private:
|
|
const string_container &str;
|
|
const size_t batch_size;
|
|
document::stream *stream{nullptr};
|
|
}; // end of class JsonStream
|
|
|
|
} // end of namespace simdjson
|
|
|
|
#endif // SIMDJSON_JSONSTREAM_H
|