mirror of
https://github.com/nlohmann/json.git
synced 2026-09-30 03:30:31 +00:00
Compare commits
14
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
448c9dd680 | ||
|
|
fa58c6e6da | ||
|
|
58706fdba0 | ||
|
|
50f0a1b75b | ||
|
|
f72abf8420 | ||
|
|
2b5e58386d | ||
|
|
95c2fe08a6 | ||
|
|
0d59ae5d49 | ||
|
|
61b104606d | ||
|
|
57b5d56522 | ||
|
|
ebfbbf29cb | ||
|
|
a491cc7663 | ||
|
|
6536c1b869 | ||
|
|
038f448dec |
@@ -52,6 +52,7 @@ labels:
|
||||
- "single_include/nlohmann/json_view\\.hpp"
|
||||
- "tests/src/unit-json_view.*"
|
||||
- "tests/src/fuzzer-parse_json_view\\.cpp"
|
||||
- "tests/benchmarks/json_view/.*"
|
||||
- "tools/amalgamate/config_json_view\\.json"
|
||||
- "docs/mkdocs/docs/features/json_view\\.md"
|
||||
- "docs/mkdocs/docs/api/basic_json_(document|view)/.*"
|
||||
|
||||
@@ -0,0 +1,78 @@
|
||||
name: "json_view benchmarks"
|
||||
|
||||
# On demand only: runs the comparison of json_view with yyjson, simdjson, and
|
||||
# Boost.JSON (tests/benchmarks/json_view/compare.py) on GitHub-hosted runners,
|
||||
# for numbers from x86-64 and AArch64 Linux. It runs when started by hand, or
|
||||
# when a pull request gets the label "benchmark" (on both architectures, with
|
||||
# GCC and the default settings). Shared runners are noisy: the results show
|
||||
# where json_view stands, but published numbers need a quiet machine (see
|
||||
# tests/benchmarks/json_view/README.md).
|
||||
|
||||
on:
|
||||
pull_request:
|
||||
types: [labeled]
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
runner:
|
||||
description: "Runner image"
|
||||
type: choice
|
||||
options:
|
||||
- ubuntu-24.04
|
||||
- ubuntu-24.04-arm
|
||||
default: ubuntu-24.04
|
||||
compiler:
|
||||
description: "Compiler"
|
||||
type: choice
|
||||
options:
|
||||
- g++
|
||||
- clang++
|
||||
default: g++
|
||||
native:
|
||||
description: "Compile for the runner's CPU (-march=native)"
|
||||
type: boolean
|
||||
default: false
|
||||
rounds:
|
||||
description: "Rounds of bench_view"
|
||||
type: number
|
||||
default: 30
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
compare:
|
||||
if: github.event_name == 'workflow_dispatch' || github.event.label.name == 'benchmark'
|
||||
strategy:
|
||||
matrix:
|
||||
runner: ${{ fromJSON(github.event_name == 'workflow_dispatch' && format('["{0}"]', inputs.runner) || '["ubuntu-24.04", "ubuntu-24.04-arm"]') }}
|
||||
runs-on: ${{ matrix.runner }}
|
||||
steps:
|
||||
- name: Harden Runner
|
||||
uses: step-security/harden-runner@e14015d583714f6e62063499dc959a02595150a1 # v2.21.1
|
||||
with:
|
||||
egress-policy: audit
|
||||
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
with:
|
||||
persist-credentials: false
|
||||
|
||||
- name: Download test data
|
||||
run: |
|
||||
cmake -S . -B build -DJSON_BuildTests=On
|
||||
cmake --build build --target download_test_data
|
||||
|
||||
- name: Run the comparison
|
||||
env:
|
||||
CXX: ${{ inputs.compiler || 'g++' }}
|
||||
CC: ${{ inputs.compiler == 'clang++' && 'clang' || 'gcc' }}
|
||||
ROUNDS: ${{ inputs.rounds || 30 }}
|
||||
NATIVE: ${{ inputs.native && '--native' || '' }}
|
||||
run: python3 tests/benchmarks/json_view/compare.py --data build/test_files --download --rounds "$ROUNDS" $NATIVE
|
||||
|
||||
- name: Summary
|
||||
run: cat tests/benchmarks/json_view/results/*.md >> "$GITHUB_STEP_SUMMARY"
|
||||
|
||||
- uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
|
||||
with:
|
||||
name: json_view-benchmarks-${{ matrix.runner }}-${{ inputs.compiler || 'g++' }}
|
||||
path: tests/benchmarks/json_view/results/
|
||||
@@ -78,9 +78,11 @@ cc_library(
|
||||
"include/nlohmann/detail/view/materialize.hpp",
|
||||
"include/nlohmann/detail/view/node.hpp",
|
||||
"include/nlohmann/detail/view/number.hpp",
|
||||
"include/nlohmann/detail/view/object_index.hpp",
|
||||
"include/nlohmann/detail/view/pointer.hpp",
|
||||
"include/nlohmann/detail/view/scan.hpp",
|
||||
"include/nlohmann/detail/view/serializer.hpp",
|
||||
"include/nlohmann/detail/view/simd.hpp",
|
||||
"include/nlohmann/detail/view/string_ref.hpp",
|
||||
"include/nlohmann/detail/view/value.hpp",
|
||||
"include/nlohmann/json.hpp",
|
||||
|
||||
@@ -1405,6 +1405,7 @@ THE SOFTWARE IS PROVIDED “AS IS”, WITHOUT WARRANTY OF ANY KIND, EXPRESS OR I
|
||||
- The class contains parts of [Google Abseil](https://github.com/abseil/abseil-cpp) which is licensed under the [Apache 2.0 License](https://opensource.org/licenses/Apache-2.0).
|
||||
- The class contains an adapted version of the Eisel-Lemire algorithm and its table of powers of five from [fast_float](https://github.com/fastfloat/fast_float) by Daniel Lemire and contributors, which is available under the [MIT License](https://opensource.org/licenses/MIT) (used here), the Apache 2.0 License, and the Boost Software License. Copyright © 2021 The fast_float authors
|
||||
- The view's parser (`<nlohmann/json_view.hpp>`) contains techniques and code adapted from [yyjson](https://github.com/ibireme/yyjson) by YaoYuan, which is licensed under the [MIT License](https://opensource.org/licenses/MIT) (see above): table-driven decoding of `\u` escapes and fixed-offset unrolled checks.
|
||||
- The view's parser (`<nlohmann/json_view.hpp>`) validates non-ASCII strings with the vector UTF-8 check of [simdjson](https://github.com/simdjson/simdjson) by Daniel Lemire, Geoff Langdale, John Keiser, and contributors (its "lookup4" algorithm and tables, after J. Keiser and D. Lemire, "Validating UTF-8 In Less Than One Instruction Per Byte", 2021), which is available under the [MIT License](https://opensource.org/licenses/MIT) (used here) and the Apache 2.0 License. Copyright © 2018-2025 The simdjson authors
|
||||
|
||||
<img align="right" src="https://git.fsfe.org/reuse/reuse-ci/raw/branch/master/reuse-horizontal.png" alt="REUSE Software">
|
||||
|
||||
|
||||
@@ -307,3 +307,5 @@ INSERT INTO searchIndex(name, type, path) VALUES ('NLOHMANN_JSON_SERIALIZE_ENUM'
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('NLOHMANN_JSON_VERSION_MAJOR', 'Macro', 'api/macros/nlohmann_json_version_major/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('NLOHMANN_JSON_VERSION_MINOR', 'Macro', 'api/macros/nlohmann_json_version_major/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('NLOHMANN_JSON_VERSION_PATCH', 'Macro', 'api/macros/nlohmann_json_version_major/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('JSON_VIEW_NO_SIMD', 'Macro', 'api/macros/json_view_no_simd/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('JSON_VIEW_USE_SSSE3', 'Macro', 'api/macros/json_view_use_ssse3/index.html');
|
||||
|
||||
@@ -74,6 +74,8 @@ None of these exceptions carry a [`JSON_DIAGNOSTICS`](../macros/json_diagnostics
|
||||
1. Linear in the number of members: as for [`ordered_json`](../ordered_json.md), members are compared one after
|
||||
another, in document order, stopping at the first match. Each comparison first checks the key's length --
|
||||
already known from the index, without reading the key bytes -- before comparing its content.
|
||||
Objects with 128 or more members get a hash index while parsing, so that a lookup in them takes constant time
|
||||
on average.
|
||||
2. Linear in `idx`: elements are skipped one at a time from the first one, since they are not a fixed size in the
|
||||
index (unlike `BasicJsonType`'s array, which is random-access).
|
||||
3. Linear in the number of reference tokens of `ptr` and, for each token, in the number of members of the object at
|
||||
|
||||
@@ -35,6 +35,8 @@ No-throw guarantee: this function never throws exceptions.
|
||||
1. Linear in the number of members: as for [`ordered_json`](../ordered_json.md), members are compared one after
|
||||
another, in document order, stopping at the first match. Each comparison first checks the key's length -- already
|
||||
known from the index, without reading the key bytes -- before comparing its content.
|
||||
Objects with 128 or more members get a hash index while parsing, so that a lookup in them takes constant time
|
||||
on average.
|
||||
2. Linear in the number of reference tokens of `ptr` and, for each token, in the number of members of the object at
|
||||
that level or the index into the array -- as for [`operator[]`](operator[].md#complexity) and
|
||||
[`at`](at.md#complexity) with a JSON pointer.
|
||||
|
||||
@@ -26,6 +26,8 @@ No-throw guarantee: this function never throws exceptions.
|
||||
Linear in the number of members: as for [`ordered_json`](../ordered_json.md), members are compared one after
|
||||
another, in document order, stopping at the first match. Each comparison first checks the key's length -- already
|
||||
known from the index, without reading the key bytes -- before comparing its content.
|
||||
Objects with 128 or more members get a hash index while parsing, so that a lookup in them takes constant time on
|
||||
average.
|
||||
|
||||
## Notes
|
||||
|
||||
|
||||
@@ -28,6 +28,8 @@ No-throw guarantee: this function never throws exceptions.
|
||||
Linear in the number of members: as for [`ordered_json`](../ordered_json.md), members are compared one after
|
||||
another, in document order, stopping at the first match. Each comparison first checks the key's length -- already
|
||||
known from the index, without reading the key bytes -- before comparing its content.
|
||||
Objects with 128 or more members get a hash index while parsing, so that a lookup in them takes constant time on
|
||||
average.
|
||||
|
||||
## Notes
|
||||
|
||||
|
||||
@@ -76,6 +76,8 @@ None of these exceptions carry a [`JSON_DIAGNOSTICS`](../macros/json_diagnostics
|
||||
another, in document order, stopping at the first match. Each comparison first checks the key's length --
|
||||
already known from the index, without reading the key bytes -- before comparing its content, so a key of a
|
||||
different length than `key` is rejected without touching the source text.
|
||||
Objects with 128 or more members get a hash index while parsing, so that a lookup in them takes constant time
|
||||
on average.
|
||||
2. Linear in `idx`: elements are skipped one at a time from the first one, since they are not a fixed size in the
|
||||
index (unlike `BasicJsonType`'s array, which is random-access).
|
||||
3. Linear in the number of reference tokens of `ptr` and, for each token, in the number of members of the object at
|
||||
|
||||
@@ -70,6 +70,8 @@ None of these exceptions carry a [`JSON_DIAGNOSTICS`](../macros/json_diagnostics
|
||||
1. Linear in the number of members: as for [`operator[]`](operator[].md#complexity), members are compared one after
|
||||
another, in document order, stopping at the first match. Plus the complexity of converting the found member to
|
||||
`T` (see [`get`](get.md)).
|
||||
Objects with 128 or more members get a hash index while parsing, so that a lookup in them takes constant time
|
||||
on average.
|
||||
2. Linear in the number of reference tokens of `ptr` and, for each token, in the number of members of the object at
|
||||
that level or the index into the array -- as for the [`operator[]`](operator[].md#complexity) and
|
||||
[`at`](at.md#complexity) overloads that take a JSON pointer. Plus the complexity of converting the resolved value
|
||||
|
||||
@@ -33,6 +33,8 @@ header. See also the [macro overview page](../../features/macros.md).
|
||||
- [**JSON_SKIP_UNSUPPORTED_COMPILER_CHECK**](json_skip_unsupported_compiler_check.md) - do not warn about unsupported compilers
|
||||
- [**JSON_USE_GLOBAL_UDLS**](json_use_global_udls.md) - place user-defined string literals (UDLs) into the global namespace
|
||||
- [**JSON_USE_SIMDUTF**](json_use_simdutf.md) - use the simdutf library to accelerate UTF-8 validation
|
||||
- [**JSON_VIEW_NO_SIMD**](json_view_no_simd.md) - use only portable code in the parser of `json_view.hpp`
|
||||
- [**JSON_VIEW_USE_SSSE3**](json_view_use_ssse3.md) - validate non-ASCII strings with SSSE3 in the parser of `json_view.hpp`
|
||||
|
||||
## Library version
|
||||
|
||||
|
||||
@@ -0,0 +1,50 @@
|
||||
# JSON_VIEW_NO_SIMD
|
||||
|
||||
```cpp
|
||||
#define JSON_VIEW_NO_SIMD
|
||||
```
|
||||
|
||||
When defined, the parser of [`basic_json_document`](../basic_json_document/index.md) (`<nlohmann/json_view.hpp>`)
|
||||
uses only portable C++ to scan strings. By default, it scans long runs of string bytes 16 at a time with NEON on
|
||||
AArch64 (with GCC and Clang) and SSE2 on x86-64, which are part of the baseline instruction sets of these
|
||||
architectures, and validates non-ASCII text with NEON (or SSSE3, see
|
||||
[`JSON_VIEW_USE_SSSE3`](json_view_use_ssse3.md)).
|
||||
|
||||
The same input is accepted or rejected either way, with the same values, and errors are reported the same way; only
|
||||
the speed differs. The macro exists for platforms whose compilers lack the intrinsics headers, and to test the portable
|
||||
code.
|
||||
|
||||
!!! warning "Define consistently"
|
||||
|
||||
The macro selects between two definitions of the same inline functions. It must therefore be defined identically for
|
||||
**every** translation unit that includes `<nlohmann/json_view.hpp>`; prefer a compile definition on the target.
|
||||
|
||||
## Default definition
|
||||
|
||||
By default, `#!cpp JSON_VIEW_NO_SIMD` is not defined, and the vector code is used where available.
|
||||
|
||||
```cpp
|
||||
#undef JSON_VIEW_NO_SIMD
|
||||
```
|
||||
|
||||
## Examples
|
||||
|
||||
??? example
|
||||
|
||||
The code below uses the portable string scanning of the view.
|
||||
|
||||
```cpp
|
||||
#define JSON_VIEW_NO_SIMD
|
||||
#include <nlohmann/json_view.hpp>
|
||||
|
||||
...
|
||||
```
|
||||
|
||||
## See also
|
||||
|
||||
- [JSON_VIEW_USE_SSSE3](json_view_use_ssse3.md) - validate non-ASCII strings with SSSE3 on x86-64
|
||||
- [json_view](../../features/json_view.md) - the zero-copy view
|
||||
|
||||
## Version history
|
||||
|
||||
- Added in version 3.13.0.
|
||||
@@ -0,0 +1,49 @@
|
||||
# JSON_VIEW_USE_SSSE3
|
||||
|
||||
```cpp
|
||||
#define JSON_VIEW_USE_SSSE3
|
||||
```
|
||||
|
||||
When defined on x86-64, the parser of [`basic_json_document`](../basic_json_document/index.md)
|
||||
(`<nlohmann/json_view.hpp>`) validates non-ASCII text in strings with SSSE3, 16 bytes at a time, using the "lookup4"
|
||||
algorithm of [simdjson](https://github.com/simdjson/simdjson). Without it, non-ASCII text is validated one UTF-8
|
||||
sequence at a time on x86-64; on AArch64, the vector check uses NEON and is always on.
|
||||
|
||||
SSSE3 is not part of the x86-64 baseline, so the code must be compiled for it: define the macro only together with a
|
||||
compiler option that enables SSSE3 (e.g. `-mssse3`, or `-march=` with a CPU that has it), and only for programs that
|
||||
run on such CPUs. The same input is accepted or rejected either way; only the speed of non-ASCII text differs.
|
||||
|
||||
!!! warning "Define consistently"
|
||||
|
||||
The macro selects between two definitions of the same inline functions. It must therefore be defined identically,
|
||||
with the same compiler options, for **every** translation unit that includes `<nlohmann/json_view.hpp>`; mixing
|
||||
translation units that define it with ones that do not is an ODR violation. Prefer a compile definition on the
|
||||
target.
|
||||
|
||||
## Default definition
|
||||
|
||||
By default, `#!cpp JSON_VIEW_USE_SSSE3` is not defined.
|
||||
|
||||
```cpp
|
||||
#undef JSON_VIEW_USE_SSSE3
|
||||
```
|
||||
|
||||
## Examples
|
||||
|
||||
??? example
|
||||
|
||||
With CMake, for a program that only runs on CPUs with SSSE3:
|
||||
|
||||
```cmake
|
||||
target_compile_definitions(your_target PRIVATE JSON_VIEW_USE_SSSE3)
|
||||
target_compile_options(your_target PRIVATE -mssse3)
|
||||
```
|
||||
|
||||
## See also
|
||||
|
||||
- [JSON_VIEW_NO_SIMD](json_view_no_simd.md) - use only portable code in the view's parser
|
||||
- [JSON_USE_SIMDUTF](json_use_simdutf.md) - validate UTF-8 with simdutf in `basic_json`'s parser
|
||||
|
||||
## Version history
|
||||
|
||||
- Added in version 3.13.0.
|
||||
@@ -196,7 +196,7 @@ packet-beta
|
||||
|-------|---------|------------|-------------------------------------------------------------------------------------------------------------------------------|
|
||||
| 0 | `kind` | `uint8_t` | the type, numbered as [`value_t`](../api/basic_json/value_t.md): 0 null, 1 object, 2 array, 3 string, 4 boolean, 5 signed integer, 6 unsigned integer, 7 float |
|
||||
| 1 | `flags` | `uint8_t` | bits 0-1: where a string's bytes are (0: the source text, 1: the buffer of decoded strings, for strings with escapes); bit 2: the value of a boolean |
|
||||
| 2-3 | `extra` | `uint16_t` | numbers: the number of integer digits (low byte) and fraction digits (high byte), 255 for more; otherwise 0 |
|
||||
| 2-3 | `extra` | `uint16_t` | numbers: the number of integer digits (low byte) and fraction digits (high byte), 255 for more; objects: the number of their hash index (1-based), or 0; otherwise 0 |
|
||||
| 4-7 | `off` | `uint32_t` | where the value starts: the first byte after a string's opening quote (or its position in the buffer of decoded strings), the first byte of a number or literal, the bracket of an array or object |
|
||||
| 8-11 | `len` | `uint32_t` | strings: the length after decoding; floats and literals: the length of the token; arrays and objects: the number of elements |
|
||||
| 12-15 | `next` | `uint32_t` | arrays and objects: the number of nodes of the subtree, including the node itself |
|
||||
@@ -211,6 +211,11 @@ packet-beta
|
||||
subtree is `next` nodes further for an array or object, and the next node otherwise (`document_data::after`). Views
|
||||
step from element to element this way and skip whole subtrees in constant time.
|
||||
- **Offsets** are 32 bits wide, so a document is limited to 4 GiB (`out_of_range.416`).
|
||||
- **Large objects** (128 members or more) get a hash index after parsing
|
||||
([`detail/view/object_index.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/view/object_index.hpp)):
|
||||
an open-addressing table whose slots hold the distance from the object's node to a key's node, so that a lookup does
|
||||
not compare every key. The object's `extra` holds the number of its table. Only 65,535 tables fit into `extra`;
|
||||
objects beyond them are searched linearly.
|
||||
|
||||
For example, `#!json {"a": [1, 2.5]}` becomes five nodes. Each node's elements follow it, and `next` leads from an
|
||||
array or object past its subtree:
|
||||
|
||||
@@ -23,3 +23,5 @@ The class contains a copy of [Hedley](https://nemequ.github.io/hedley/) from Eva
|
||||
The class contains an adapted version of the Eisel-Lemire algorithm and its table of powers of five from [fast_float](https://github.com/fastfloat/fast_float) by Daniel Lemire and contributors, which is available under the [MIT License](https://opensource.org/licenses/MIT) (used here), the Apache 2.0 License, and the Boost Software License. Copyright © 2021 The fast_float authors
|
||||
|
||||
The view's parser (`<nlohmann/json_view.hpp>`) contains techniques and code adapted from [yyjson](https://github.com/ibireme/yyjson) by YaoYuan, which is licensed under the [MIT License](https://opensource.org/licenses/MIT) (see above): table-driven decoding of `\u` escapes and fixed-offset unrolled checks.
|
||||
|
||||
The view's parser (`<nlohmann/json_view.hpp>`) validates non-ASCII strings with the vector UTF-8 check of [simdjson](https://github.com/simdjson/simdjson) by Daniel Lemire, Geoff Langdale, John Keiser, and contributors (its "lookup4" algorithm and tables, after J. Keiser and D. Lemire, "Validating UTF-8 In Less Than One Instruction Per Byte", 2021), which is available under the [MIT License](https://opensource.org/licenses/MIT) (used here) and the Apache 2.0 License. Copyright © 2018-2025 The simdjson authors
|
||||
|
||||
@@ -368,6 +368,8 @@ nav:
|
||||
- 'JSON_USE_IMPLICIT_CONVERSIONS': api/macros/json_use_implicit_conversions.md
|
||||
- 'JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON': api/macros/json_use_legacy_discarded_value_comparison.md
|
||||
- 'JSON_USE_SIMDUTF': api/macros/json_use_simdutf.md
|
||||
- 'JSON_VIEW_NO_SIMD': api/macros/json_view_no_simd.md
|
||||
- 'JSON_VIEW_USE_SSSE3': api/macros/json_view_use_ssse3.md
|
||||
- 'NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE, NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE_WITH_DEFAULT, NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE_ONLY_SERIALIZE, NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE, NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_WITH_DEFAULT, NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_ONLY_SERIALIZE': api/macros/nlohmann_define_derived_type.md
|
||||
- 'NLOHMANN_DEFINE_TYPE_INTRUSIVE, NLOHMANN_DEFINE_TYPE_INTRUSIVE_WITH_DEFAULT, NLOHMANN_DEFINE_TYPE_INTRUSIVE_ONLY_SERIALIZE': api/macros/nlohmann_define_type_intrusive.md
|
||||
- 'NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE, NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE_WITH_DEFAULT, NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE_ONLY_SERIALIZE': api/macros/nlohmann_define_type_non_intrusive.md
|
||||
|
||||
@@ -115,6 +115,13 @@ class builder
|
||||
frame shallow[64]; // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays): not initialized on purpose; filled as containers open
|
||||
std::vector<frame> deep{};
|
||||
|
||||
/// remember an object to index after parsing (out of line, so that the
|
||||
/// parse loop only has a call for it)
|
||||
NLOHMANN_VIEW_NOINLINE void note_large_object(std::uint32_t idx)
|
||||
{
|
||||
doc.large_objects.push_back(idx);
|
||||
}
|
||||
|
||||
NLOHMANN_VIEW_NOINLINE bool fail(error_code c, const unsigned char* at) noexcept
|
||||
{
|
||||
m_failure.code = c;
|
||||
@@ -501,7 +508,7 @@ class builder
|
||||
switch (cur()) \
|
||||
{ \
|
||||
case '"': \
|
||||
if (NLOHMANN_VIEW_UNLIKELY(!string())) { return false; } \
|
||||
if (NLOHMANN_VIEW_UNLIKELY(!string<true>())) { return false; } \
|
||||
goto NEXT; \
|
||||
case '{': \
|
||||
open(value_t::object); \
|
||||
@@ -585,7 +592,7 @@ obj_key:
|
||||
{
|
||||
return fail(error_code::expected_key);
|
||||
}
|
||||
if (NLOHMANN_VIEW_UNLIKELY(!string()))
|
||||
if (NLOHMANN_VIEW_UNLIKELY(!string<false>()))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
@@ -626,19 +633,27 @@ obj_next:
|
||||
if (enabled(TrailingCommas) && cur() == '}')
|
||||
{
|
||||
++p;
|
||||
goto close_container;
|
||||
goto close_object;
|
||||
}
|
||||
goto obj_key;
|
||||
}
|
||||
if (cur() == '}')
|
||||
{
|
||||
++p;
|
||||
goto close_container;
|
||||
goto close_object;
|
||||
}
|
||||
return fail(error_code::expected_object_end);
|
||||
|
||||
#undef NLOHMANN_VIEW_VALUE
|
||||
|
||||
close_object:
|
||||
// a large object gets a hash index (objects only, so that closing
|
||||
// an array pays nothing for this)
|
||||
if (NLOHMANN_VIEW_UNLIKELY(cur_count >= document_data::index_min_members))
|
||||
{
|
||||
cold.note_large_object(cur_idx);
|
||||
}
|
||||
|
||||
close_container:
|
||||
close();
|
||||
if (NLOHMANN_VIEW_UNLIKELY(depth == 0))
|
||||
@@ -688,7 +703,7 @@ root_done:
|
||||
switch (cur())
|
||||
{
|
||||
case '"':
|
||||
return string();
|
||||
return string<true>();
|
||||
case 't':
|
||||
return literal("true", 4, value_t::boolean, node_flags::is_true);
|
||||
case 'f':
|
||||
@@ -974,12 +989,13 @@ indent_done:
|
||||
return true;
|
||||
}
|
||||
|
||||
/// a string (value or key) at p
|
||||
/// a string at p: a value (Value) or a key
|
||||
template<bool Value>
|
||||
NLOHMANN_VIEW_ALWAYS_INLINE bool string()
|
||||
{
|
||||
++p; // opening quote
|
||||
const unsigned char* const s = p;
|
||||
p = scan_string_run(p, e);
|
||||
p = scan_string_run<Value>(p, e);
|
||||
if (NLOHMANN_VIEW_LIKELY(p != e && *p == '"'))
|
||||
{
|
||||
emit(value_t::string, 0, 0, static_cast<std::size_t>(s - b), static_cast<std::uint64_t>(p - s));
|
||||
|
||||
@@ -10,9 +10,11 @@
|
||||
|
||||
#include <array> // array
|
||||
#include <cstddef> // size_t
|
||||
#include <cstdint> // uint32_t
|
||||
#include <cstring> // memcpy
|
||||
#include <new> // operator new, placement new
|
||||
#include <string> // string
|
||||
#include <vector> // vector
|
||||
|
||||
#include <nlohmann/json.hpp>
|
||||
#include <nlohmann/detail/view/macro_scope.hpp>
|
||||
@@ -37,6 +39,17 @@ struct document_data
|
||||
std::size_t inline_cap = 0;
|
||||
std::string arena{}; ///< decoded strings that contained escapes // NOLINT(readability-redundant-member-init)
|
||||
std::string owned{}; ///< owned copy of the input, if any // NOLINT(readability-redundant-member-init)
|
||||
|
||||
// hash indexes of large objects (see object_index.hpp)
|
||||
static constexpr std::uint32_t index_min_members = 128;
|
||||
struct object_index
|
||||
{
|
||||
std::size_t start; ///< first slot in index_slots
|
||||
std::uint32_t mask; ///< slot count - 1 (a power of two minus one)
|
||||
};
|
||||
std::vector<object_index> indexes{}; // NOLINT(readability-redundant-member-init)
|
||||
std::vector<std::uint32_t> index_slots{}; // NOLINT(readability-redundant-member-init)
|
||||
std::vector<std::uint32_t> large_objects{}; ///< positions of the objects to index (noted while parsing) // NOLINT(readability-redundant-member-init)
|
||||
std::array<const char*, 4> base = {{nullptr, nullptr, nullptr, nullptr}}; ///< string bases: source, arena (indexed by flags & node_flags::storage)
|
||||
bool discarded = true;
|
||||
|
||||
@@ -64,7 +77,7 @@ struct document_data
|
||||
}
|
||||
};
|
||||
|
||||
document_data() noexcept = default;
|
||||
document_data() = default;
|
||||
document_data(const document_data&) = delete;
|
||||
document_data(document_data&&) = delete;
|
||||
document_data& operator=(const document_data&) = delete;
|
||||
|
||||
@@ -16,6 +16,7 @@
|
||||
#include <nlohmann/detail/view/document_data.hpp>
|
||||
#include <nlohmann/detail/view/macro_scope.hpp>
|
||||
#include <nlohmann/detail/view/node.hpp>
|
||||
#include <nlohmann/detail/view/object_index.hpp>
|
||||
|
||||
NLOHMANN_JSON_NAMESPACE_BEGIN
|
||||
namespace detail
|
||||
@@ -85,6 +86,10 @@ class short_key
|
||||
/// nullptr; most keys are rejected by their length, from the index alone
|
||||
inline const node* find_member(const document_data& d, const node* object, const char* key, std::size_t n) noexcept
|
||||
{
|
||||
if (NLOHMANN_VIEW_UNLIKELY(object->extra != 0))
|
||||
{
|
||||
return find_indexed(d, object, key, n); // a large object
|
||||
}
|
||||
const node* const end = document_data::child_end(object);
|
||||
const auto* const k = reinterpret_cast<const unsigned char*>(key); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
|
||||
if (NLOHMANN_VIEW_LIKELY(n <= 16))
|
||||
|
||||
@@ -19,3 +19,8 @@
|
||||
#undef NLOHMANN_VIEW_THROW
|
||||
#undef NLOHMANN_VIEW_LITTLE_ENDIAN
|
||||
#undef NLOHMANN_VIEW_REPEAT16
|
||||
#undef NLOHMANN_VIEW_NEON
|
||||
#undef NLOHMANN_VIEW_SSE2
|
||||
#undef NLOHMANN_VIEW_SSSE3
|
||||
#undef NLOHMANN_VIEW_VECTOR
|
||||
#undef NLOHMANN_VIEW_VECTOR_UTF8
|
||||
|
||||
@@ -0,0 +1,131 @@
|
||||
// __ _____ _____ _____
|
||||
// __| | __| | | | JSON for Modern C++
|
||||
// | | |__ | | | | | | version 3.12.0
|
||||
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
|
||||
//
|
||||
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <cstddef> // size_t
|
||||
#include <cstdint> // uint32_t, uint64_t
|
||||
#include <cstring> // memcmp
|
||||
|
||||
#include <nlohmann/json.hpp>
|
||||
#include <nlohmann/detail/view/document_data.hpp>
|
||||
#include <nlohmann/detail/view/macro_scope.hpp>
|
||||
#include <nlohmann/detail/view/node.hpp>
|
||||
|
||||
// Hash indexes of large objects, so that a lookup does not compare thousands
|
||||
// of keys (as Boost.JSON switches from a linear search to a hash table for
|
||||
// large objects). An object with document_data::index_min_members members or
|
||||
// more gets an open-addressing table after parsing; its node stores the
|
||||
// number of the table (1-based) in `extra`. A slot holds the offset of a key
|
||||
// node from its object node (0: empty). Of duplicate keys, the first is kept,
|
||||
// as for the linear search.
|
||||
|
||||
NLOHMANN_JSON_NAMESPACE_BEGIN
|
||||
namespace detail
|
||||
{
|
||||
namespace view
|
||||
{
|
||||
|
||||
/// hash of a key: its bytes, eight at a time, in a fixed byte order
|
||||
inline std::uint64_t key_hash(const char* s, std::size_t n) noexcept
|
||||
{
|
||||
std::uint64_t h = 0x9E3779B97F4A7C15u * (n + 1);
|
||||
const auto* p = reinterpret_cast<const unsigned char*>(s); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
|
||||
while (n >= 8)
|
||||
{
|
||||
h = (h ^ read_eight_bytes(p)) * 0xBF58476D1CE4E5B9u;
|
||||
h ^= h >> 29u;
|
||||
p += 8;
|
||||
n -= 8;
|
||||
}
|
||||
std::uint64_t w = 0;
|
||||
for (std::size_t i = 0; i < n; ++i)
|
||||
{
|
||||
w |= static_cast<std::uint64_t>(p[i]) << (8u * i);
|
||||
}
|
||||
h = (h ^ w) * 0x94D049BB133111EBu;
|
||||
return h ^ (h >> 31u);
|
||||
}
|
||||
|
||||
/// build the table of a large object
|
||||
inline void build_object_index(document_data& d, node* obj)
|
||||
{
|
||||
if (d.indexes.size() >= 0xFFFFu)
|
||||
{
|
||||
return; // LCOV_EXCL_LINE (the number must fit `extra`; more large objects are searched linearly)
|
||||
}
|
||||
std::size_t cap = 16;
|
||||
while (cap < 2 * static_cast<std::size_t>(obj->len))
|
||||
{
|
||||
cap *= 2;
|
||||
}
|
||||
const std::size_t start = d.index_slots.size();
|
||||
d.index_slots.resize(start + cap, 0);
|
||||
std::uint32_t* const slots = d.index_slots.data() + start;
|
||||
const std::size_t mask = cap - 1;
|
||||
for (const node* k = document_data::first_child(obj), *end = document_data::child_end(obj); k != end; k = document_data::after(k + 1))
|
||||
{
|
||||
const char* const key = d.str(*k);
|
||||
const std::uint64_t hash = key_hash(key, k->len); // (a cast of the call would be useless where std::uint64_t is std::size_t)
|
||||
std::size_t i = static_cast<std::size_t>(hash) & mask;
|
||||
bool duplicate = false;
|
||||
while (slots[i] != 0)
|
||||
{
|
||||
const node* const other = obj + slots[i];
|
||||
if (other->len == k->len && (k->len == 0 || std::memcmp(d.str(*other), key, k->len) == 0))
|
||||
{
|
||||
duplicate = true; // keep the first
|
||||
break;
|
||||
}
|
||||
i = (i + 1) & mask;
|
||||
}
|
||||
if (!duplicate)
|
||||
{
|
||||
slots[i] = static_cast<std::uint32_t>(k - obj);
|
||||
}
|
||||
}
|
||||
d.indexes.push_back(document_data::object_index{start, static_cast<std::uint32_t>(mask)});
|
||||
obj->extra = static_cast<std::uint16_t>(d.indexes.size());
|
||||
}
|
||||
|
||||
/// build the tables of the large objects the parser noted
|
||||
inline void build_object_indexes(document_data& d)
|
||||
{
|
||||
for (const std::uint32_t i : d.large_objects)
|
||||
{
|
||||
build_object_index(d, d.tape + i);
|
||||
}
|
||||
}
|
||||
|
||||
/// the key node of the first member with this key of an indexed object, or
|
||||
/// nullptr
|
||||
inline const node* find_indexed(const document_data& d, const node* obj, const char* key, std::size_t n) noexcept
|
||||
{
|
||||
const document_data::object_index& ix = d.indexes[obj->extra - 1u];
|
||||
const std::uint32_t* const slots = d.index_slots.data() + ix.start;
|
||||
const std::uint64_t hash = key_hash(key, n); // (a cast of the call would be useless where std::uint64_t is std::size_t)
|
||||
std::size_t i = static_cast<std::size_t>(hash) & ix.mask;
|
||||
for (;;)
|
||||
{
|
||||
const std::uint32_t s = slots[i];
|
||||
if (s == 0)
|
||||
{
|
||||
return nullptr;
|
||||
}
|
||||
const node* const k = obj + s;
|
||||
if (k->len == n && (n == 0 || std::memcmp(d.str(*k), key, n) == 0))
|
||||
{
|
||||
return k;
|
||||
}
|
||||
i = (i + 1) & ix.mask;
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace view
|
||||
} // namespace detail
|
||||
NLOHMANN_JSON_NAMESPACE_END
|
||||
@@ -16,6 +16,7 @@
|
||||
|
||||
#include <nlohmann/json.hpp>
|
||||
#include <nlohmann/detail/view/macro_scope.hpp>
|
||||
#include <nlohmann/detail/view/simd.hpp>
|
||||
|
||||
// Scanning primitives of the view's parser. The unrolled checks at fixed
|
||||
// offsets follow yyjson (https://github.com/ibireme/yyjson, MIT license): the
|
||||
@@ -62,9 +63,12 @@ NLOHMANN_VIEW_ALWAYS_INLINE std::uint16_t load16(const unsigned char* p) noexcep
|
||||
|
||||
/// Advance over plain string bytes and well-formed UTF-8. Stops at a quote,
|
||||
/// a backslash, a control character, ill-formed UTF-8, or the end. The first
|
||||
/// 16 bytes are checked one by one, so that the position advances by
|
||||
/// constants in predicted branches (most strings are short); longer runs
|
||||
/// continue eight bytes at a time.
|
||||
/// bytes are checked one by one, so that the position advances by constants
|
||||
/// in predicted branches: 16 for keys, whose lengths repeat from record to
|
||||
/// record, and 8 for string values (Value) where a vector loop follows, as
|
||||
/// their lengths vary more. Longer runs continue 16 bytes at a time with NEON
|
||||
/// or SSE2, else eight bytes at a time.
|
||||
template<bool Value = false>
|
||||
NLOHMANN_VIEW_ALWAYS_INLINE const unsigned char* scan_string_run(const unsigned char* p, const unsigned char* e) noexcept
|
||||
{
|
||||
const std::uint8_t* plain = string_plain();
|
||||
@@ -73,9 +77,23 @@ NLOHMANN_VIEW_ALWAYS_INLINE const unsigned char* scan_string_run(const unsigned
|
||||
if (e - p >= 16)
|
||||
{
|
||||
#define NLOHMANN_VIEW_STEP(i) if (NLOHMANN_VIEW_LIKELY(plain[p[i]] != 0)) {} else { p += (i); goto stop; }
|
||||
NLOHMANN_VIEW_REPEAT16(NLOHMANN_VIEW_STEP)
|
||||
NLOHMANN_VIEW_STEP(0) NLOHMANN_VIEW_STEP(1) NLOHMANN_VIEW_STEP(2) NLOHMANN_VIEW_STEP(3)
|
||||
NLOHMANN_VIEW_STEP(4) NLOHMANN_VIEW_STEP(5) NLOHMANN_VIEW_STEP(6) NLOHMANN_VIEW_STEP(7)
|
||||
if (!Value || !NLOHMANN_VIEW_VECTOR)
|
||||
{
|
||||
NLOHMANN_VIEW_STEP(8) NLOHMANN_VIEW_STEP(9) NLOHMANN_VIEW_STEP(10) NLOHMANN_VIEW_STEP(11)
|
||||
NLOHMANN_VIEW_STEP(12) NLOHMANN_VIEW_STEP(13) NLOHMANN_VIEW_STEP(14) NLOHMANN_VIEW_STEP(15)
|
||||
p += 8;
|
||||
}
|
||||
#undef NLOHMANN_VIEW_STEP
|
||||
p += 16;
|
||||
p += 8;
|
||||
#if NLOHMANN_VIEW_VECTOR
|
||||
p = vector_plain_run(p, e);
|
||||
if (p != e && plain[*p] == 0)
|
||||
{
|
||||
goto stop;
|
||||
}
|
||||
#else
|
||||
while (e - p >= 8)
|
||||
{
|
||||
const std::uint64_t special = swar_string_special(read_eight_bytes(p));
|
||||
@@ -86,6 +104,7 @@ NLOHMANN_VIEW_ALWAYS_INLINE const unsigned char* scan_string_run(const unsigned
|
||||
}
|
||||
p += 8;
|
||||
}
|
||||
#endif
|
||||
continue;
|
||||
}
|
||||
while (p != e && plain[*p] != 0)
|
||||
@@ -101,6 +120,10 @@ stop:
|
||||
{
|
||||
return p; // quote, backslash, or control character
|
||||
}
|
||||
#if NLOHMANN_VIEW_VECTOR_UTF8
|
||||
// non-ASCII: the vector check, out of line
|
||||
return scan_string_vector(p, e, plain);
|
||||
#else
|
||||
// non-ASCII: a run of well-formed sequences (the library's check, so
|
||||
// that exactly what json::parse accepts is accepted)
|
||||
do
|
||||
@@ -113,6 +136,7 @@ stop:
|
||||
p += n;
|
||||
}
|
||||
while (p != e && *p >= 0x80);
|
||||
#endif
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,281 @@
|
||||
// __ _____ _____ _____
|
||||
// __| | __| | | | JSON for Modern C++
|
||||
// | | |__ | | | | | | version 3.12.0
|
||||
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
|
||||
//
|
||||
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
|
||||
// SPDX-FileCopyrightText: 2018-2025 The simdjson authors <https://github.com/simdjson/simdjson>
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <array> // array
|
||||
#include <cstddef> // size_t
|
||||
#include <cstdint> // uint8_t, uint64_t
|
||||
|
||||
#include <nlohmann/json.hpp>
|
||||
#include <nlohmann/detail/view/macro_scope.hpp>
|
||||
|
||||
// Vector code for long runs of string bytes. NEON (AArch64) and SSE2 (x86-64)
|
||||
// belong to the baseline instruction sets and are used by default. The vector
|
||||
// UTF-8 check needs NEON, or SSSE3 if JSON_VIEW_USE_SSSE3 is defined: SSSE3 is
|
||||
// not part of x86-64, so it must not depend on the flags of a translation unit
|
||||
// (two translation units with different flags would have different
|
||||
// definitions of the same inline functions). JSON_VIEW_NO_SIMD selects the
|
||||
// portable code.
|
||||
#if !defined(JSON_VIEW_NO_SIMD) && defined(__aarch64__) && (defined(__GNUC__) || defined(__clang__)) && NLOHMANN_VIEW_LITTLE_ENDIAN
|
||||
#include <arm_neon.h>
|
||||
#define NLOHMANN_VIEW_NEON 1
|
||||
#else
|
||||
#define NLOHMANN_VIEW_NEON 0
|
||||
#endif
|
||||
#if !defined(JSON_VIEW_NO_SIMD) && !NLOHMANN_VIEW_NEON && (defined(__SSE2__) || defined(_M_X64) || (defined(_M_IX86_FP) && _M_IX86_FP >= 2))
|
||||
#include <emmintrin.h>
|
||||
#define NLOHMANN_VIEW_SSE2 1
|
||||
#else
|
||||
#define NLOHMANN_VIEW_SSE2 0
|
||||
#endif
|
||||
#if NLOHMANN_VIEW_SSE2 && defined(JSON_VIEW_USE_SSSE3)
|
||||
#include <tmmintrin.h>
|
||||
#define NLOHMANN_VIEW_SSSE3 1 // NOLINT(cppcoreguidelines-macro-to-enum,modernize-macro-to-enum)
|
||||
#else
|
||||
#define NLOHMANN_VIEW_SSSE3 0 // NOLINT(cppcoreguidelines-macro-to-enum,modernize-macro-to-enum)
|
||||
#endif
|
||||
#define NLOHMANN_VIEW_VECTOR (NLOHMANN_VIEW_NEON || NLOHMANN_VIEW_SSE2)
|
||||
#define NLOHMANN_VIEW_VECTOR_UTF8 (NLOHMANN_VIEW_NEON || NLOHMANN_VIEW_SSSE3)
|
||||
|
||||
NLOHMANN_JSON_NAMESPACE_BEGIN
|
||||
namespace detail
|
||||
{
|
||||
namespace view
|
||||
{
|
||||
|
||||
#if NLOHMANN_VIEW_VECTOR
|
||||
/*!
|
||||
@brief the first byte of a string run that is a quote, a backslash, a control
|
||||
character, or not ASCII, 16 bytes per step
|
||||
|
||||
Stops at such a byte, or where fewer than 16 bytes are left (the caller tells
|
||||
the two apart). A signed compare with 0x20 finds control characters and
|
||||
non-ASCII bytes at once.
|
||||
*/
|
||||
NLOHMANN_VIEW_ALWAYS_INLINE const unsigned char* vector_plain_run(const unsigned char* p, const unsigned char* e) noexcept
|
||||
{
|
||||
while (e - p >= 16)
|
||||
{
|
||||
#if NLOHMANN_VIEW_NEON
|
||||
const uint8x16_t in = vld1q_u8(p);
|
||||
const uint8x16_t special = vorrq_u8(vorrq_u8(vceqq_u8(in, vdupq_n_u8('"')), vceqq_u8(in, vdupq_n_u8('\\'))),
|
||||
vcltq_s8(vreinterpretq_s8_u8(in), vdupq_n_s8(0x20)));
|
||||
// one nibble per byte (the usual NEON replacement of x86's movemask, see
|
||||
// D. Kutenin, "Porting x86 vector bitmask optimizations to Arm NEON", 2022)
|
||||
const std::uint64_t bits = vget_lane_u64(vreinterpret_u64_u8(vshrn_n_u16(vreinterpretq_u16_u8(special), 4)), 0);
|
||||
if (bits != 0)
|
||||
{
|
||||
return p + (count_trailing_zeros(bits) >> 2u);
|
||||
}
|
||||
#else
|
||||
const __m128i in = _mm_loadu_si128(static_cast<const __m128i*>(static_cast<const void*>(p)));
|
||||
const __m128i special = _mm_or_si128(_mm_or_si128(_mm_cmpeq_epi8(in, _mm_set1_epi8('"')), _mm_cmpeq_epi8(in, _mm_set1_epi8('\\'))),
|
||||
_mm_cmplt_epi8(in, _mm_set1_epi8(0x20)));
|
||||
const auto bits = static_cast<std::uint64_t>(static_cast<unsigned>(_mm_movemask_epi8(special)));
|
||||
if (bits != 0)
|
||||
{
|
||||
return p + count_trailing_zeros(bits);
|
||||
}
|
||||
#endif
|
||||
p += 16;
|
||||
}
|
||||
return p;
|
||||
}
|
||||
#endif
|
||||
|
||||
#if NLOHMANN_VIEW_VECTOR_UTF8
|
||||
/// Tables of the UTF-8 check of J. Keiser and D. Lemire, "Validating UTF-8 In
|
||||
/// Less Than One Instruction Per Byte" (2021), as in simdjson ("lookup4"): each
|
||||
/// maps a nibble (high and low nibble of the previous byte, high nibble of the
|
||||
/// current byte) to the errors it allows; a byte pair is ill-formed if all
|
||||
/// three have an error bit in common.
|
||||
template<typename Dummy = void>
|
||||
struct utf8_lookup4
|
||||
{
|
||||
static constexpr std::uint8_t too_short = 1u << 0u, too_long = 1u << 1u, overlong_3 = 1u << 2u, too_large = 1u << 3u;
|
||||
static constexpr std::uint8_t surrogate = 1u << 4u, overlong_2 = 1u << 5u, too_large_1000 = 1u << 6u, overlong_4 = 1u << 6u;
|
||||
static constexpr std::uint8_t two_conts = 1u << 7u, carry = too_short | too_long | two_conts;
|
||||
static const std::array<std::uint8_t, 16> byte_1_high;
|
||||
static const std::array<std::uint8_t, 16> byte_1_low;
|
||||
static const std::array<std::uint8_t, 16> byte_2_high;
|
||||
};
|
||||
|
||||
template<typename Dummy>
|
||||
const std::array<std::uint8_t, 16> utf8_lookup4<Dummy>::byte_1_high =
|
||||
{
|
||||
{
|
||||
too_long, too_long, too_long, too_long, too_long, too_long, too_long, too_long,
|
||||
two_conts, two_conts, two_conts, two_conts,
|
||||
too_short | overlong_2, too_short, too_short | overlong_3 | surrogate, too_short | too_large | too_large_1000 | overlong_4
|
||||
}
|
||||
};
|
||||
|
||||
template<typename Dummy>
|
||||
const std::array<std::uint8_t, 16> utf8_lookup4<Dummy>::byte_1_low =
|
||||
{
|
||||
{
|
||||
carry | overlong_3 | overlong_2 | overlong_4, carry | overlong_2, carry, carry,
|
||||
carry | too_large, carry | too_large | too_large_1000, carry | too_large | too_large_1000, carry | too_large | too_large_1000,
|
||||
carry | too_large | too_large_1000, carry | too_large | too_large_1000, carry | too_large | too_large_1000, carry | too_large | too_large_1000,
|
||||
carry | too_large | too_large_1000, carry | too_large | too_large_1000 | surrogate, carry | too_large | too_large_1000, carry | too_large | too_large_1000
|
||||
}
|
||||
};
|
||||
|
||||
template<typename Dummy>
|
||||
const std::array<std::uint8_t, 16> utf8_lookup4<Dummy>::byte_2_high =
|
||||
{
|
||||
{
|
||||
too_short, too_short, too_short, too_short, too_short, too_short, too_short, too_short,
|
||||
static_cast<std::uint8_t>(too_long | overlong_2 | two_conts | overlong_3 | too_large_1000 | overlong_4),
|
||||
static_cast<std::uint8_t>(too_long | overlong_2 | two_conts | overlong_3 | too_large),
|
||||
static_cast<std::uint8_t>(too_long | overlong_2 | two_conts | surrogate | too_large),
|
||||
static_cast<std::uint8_t>(too_long | overlong_2 | two_conts | surrogate | too_large),
|
||||
too_short, too_short, too_short, too_short
|
||||
}
|
||||
};
|
||||
|
||||
/// the end of scan_string_vector from block, where the vector loop stopped
|
||||
/// (ill-formed UTF-8, or fewer than 16 bytes left): one byte or sequence at a
|
||||
/// time, from the start of a sequence that crosses into the block
|
||||
inline const unsigned char* scan_string_finish(const unsigned char* p, const unsigned char* block, const unsigned char* e, const std::uint8_t* plain) noexcept
|
||||
{
|
||||
for (int i = 1; i <= 3 && block - i >= p; ++i)
|
||||
{
|
||||
const unsigned char c = block[-i];
|
||||
if (c < 0x80)
|
||||
{
|
||||
break;
|
||||
}
|
||||
if (c >= 0xC0)
|
||||
{
|
||||
const int len = 2 + static_cast<int>(c >= 0xE0) + static_cast<int>(c >= 0xF0);
|
||||
if (len > i)
|
||||
{
|
||||
block -= i;
|
||||
}
|
||||
break;
|
||||
}
|
||||
}
|
||||
for (p = block; p != e;)
|
||||
{
|
||||
if (*p < 0x80)
|
||||
{
|
||||
if (plain[*p] == 0)
|
||||
{
|
||||
return p;
|
||||
}
|
||||
++p;
|
||||
continue;
|
||||
}
|
||||
const std::size_t n = validate_one_utf8(p, static_cast<std::size_t>(e - p));
|
||||
if (n == 0)
|
||||
{
|
||||
return p;
|
||||
}
|
||||
p += n;
|
||||
}
|
||||
return p;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief the rest of a string from p (a character boundary), 16 bytes per step
|
||||
|
||||
The first quote, backslash, or control character is found with vector
|
||||
compares, and the UTF-8 check covers the bytes up to it. Returns where the
|
||||
string scan stops, like scan_string_run: before ill-formed UTF-8 and for the
|
||||
last bytes of the input, the bytes are checked one sequence at a time. Out of
|
||||
line, so that no constants of the check occupy registers in the parse loop.
|
||||
*/
|
||||
NLOHMANN_VIEW_NOINLINE inline const unsigned char* scan_string_vector(const unsigned char* p, const unsigned char* e, const std::uint8_t* plain) noexcept
|
||||
{
|
||||
using lookup = utf8_lookup4<>;
|
||||
const unsigned char* block = p;
|
||||
#if NLOHMANN_VIEW_NEON
|
||||
const uint8x16_t t1h = vld1q_u8(lookup::byte_1_high.data());
|
||||
const uint8x16_t t1l = vld1q_u8(lookup::byte_1_low.data());
|
||||
const uint8x16_t t2h = vld1q_u8(lookup::byte_2_high.data());
|
||||
uint8x16_t prev = vdupq_n_u8(0);
|
||||
while (e - block >= 16)
|
||||
{
|
||||
const uint8x16_t in = vld1q_u8(block);
|
||||
const uint8x16_t special = vorrq_u8(vorrq_u8(vceqq_u8(in, vdupq_n_u8('"')), vceqq_u8(in, vdupq_n_u8('\\'))), vcltq_u8(in, vdupq_n_u8(0x20)));
|
||||
const uint8x16_t prev1 = vextq_u8(prev, in, 15);
|
||||
const uint8x16_t sc = vandq_u8(vandq_u8(vqtbl1q_u8(t1h, vshrq_n_u8(prev1, 4)), vqtbl1q_u8(t1l, vandq_u8(prev1, vdupq_n_u8(0x0F)))), vqtbl1q_u8(t2h, vshrq_n_u8(in, 4)));
|
||||
const uint8x16_t must23 = vorrq_u8(vqsubq_u8(vextq_u8(prev, in, 14), vdupq_n_u8(0xE0 - 0x80)), vqsubq_u8(vextq_u8(prev, in, 13), vdupq_n_u8(0xF0 - 0x80)));
|
||||
const uint8x16_t err = veorq_u8(vandq_u8(must23, vdupq_n_u8(0x80)), sc);
|
||||
const std::uint64_t special_bits = vget_lane_u64(vreinterpret_u64_u8(vshrn_n_u16(vreinterpretq_u16_u8(special), 4)), 0);
|
||||
const std::uint64_t err_bits = vget_lane_u64(vreinterpret_u64_u8(vshrn_n_u16(vreinterpretq_u16_u8(vtstq_u8(err, err)), 4)), 0);
|
||||
if (special_bits != 0)
|
||||
{
|
||||
// errors up to the special byte count (an incomplete sequence
|
||||
// before a quote shows at the quote); the bytes after it do not
|
||||
const unsigned k = static_cast<unsigned>(count_trailing_zeros(special_bits)) >> 2u;
|
||||
const std::uint64_t upto = k == 15 ? ~std::uint64_t{0} :
|
||||
(std::uint64_t{1} << (4u * (k + 1u))) - 1u;
|
||||
if ((err_bits & upto) == 0)
|
||||
{
|
||||
return block + k;
|
||||
}
|
||||
break;
|
||||
}
|
||||
if (err_bits != 0)
|
||||
{
|
||||
break;
|
||||
}
|
||||
prev = in;
|
||||
block += 16;
|
||||
}
|
||||
#else
|
||||
// the same with SSSE3 (pshufb for the table lookups; nibbles from 16-bit
|
||||
// shifts, as there are no byte shifts)
|
||||
const __m128i t1h = _mm_loadu_si128(static_cast<const __m128i*>(static_cast<const void*>(lookup::byte_1_high.data())));
|
||||
const __m128i t1l = _mm_loadu_si128(static_cast<const __m128i*>(static_cast<const void*>(lookup::byte_1_low.data())));
|
||||
const __m128i t2h = _mm_loadu_si128(static_cast<const __m128i*>(static_cast<const void*>(lookup::byte_2_high.data())));
|
||||
const __m128i nibble = _mm_set1_epi8(0x0F);
|
||||
const __m128i zero = _mm_setzero_si128();
|
||||
__m128i prev = zero;
|
||||
while (e - block >= 16)
|
||||
{
|
||||
const __m128i in = _mm_loadu_si128(static_cast<const __m128i*>(static_cast<const void*>(block)));
|
||||
const __m128i special = _mm_or_si128(_mm_or_si128(_mm_cmpeq_epi8(in, _mm_set1_epi8('"')), _mm_cmpeq_epi8(in, _mm_set1_epi8('\\'))),
|
||||
_mm_cmpeq_epi8(_mm_subs_epu8(in, _mm_set1_epi8(0x1F)), zero)); // in < 0x20
|
||||
const __m128i prev1 = _mm_alignr_epi8(in, prev, 15);
|
||||
const __m128i sc = _mm_and_si128(_mm_and_si128(_mm_shuffle_epi8(t1h, _mm_and_si128(_mm_srli_epi16(prev1, 4), nibble)),
|
||||
_mm_shuffle_epi8(t1l, _mm_and_si128(prev1, nibble))),
|
||||
_mm_shuffle_epi8(t2h, _mm_and_si128(_mm_srli_epi16(in, 4), nibble)));
|
||||
const __m128i must23 = _mm_or_si128(_mm_subs_epu8(_mm_alignr_epi8(in, prev, 14), _mm_set1_epi8(0xE0 - 0x80)),
|
||||
_mm_subs_epu8(_mm_alignr_epi8(in, prev, 13), _mm_set1_epi8(0xF0 - 0x80)));
|
||||
const __m128i err = _mm_xor_si128(_mm_and_si128(must23, _mm_set1_epi8(static_cast<char>(-128))), sc);
|
||||
const auto special_bits = static_cast<unsigned>(_mm_movemask_epi8(special));
|
||||
const auto err_bits = ~static_cast<unsigned>(_mm_movemask_epi8(_mm_cmpeq_epi8(err, zero))) & 0xFFFFu;
|
||||
if (special_bits != 0)
|
||||
{
|
||||
const unsigned k = static_cast<unsigned>(count_trailing_zeros(static_cast<std::uint64_t>(special_bits)));
|
||||
if ((err_bits & ((2u << k) - 1u)) == 0)
|
||||
{
|
||||
return block + k;
|
||||
}
|
||||
break;
|
||||
}
|
||||
if (err_bits != 0)
|
||||
{
|
||||
break;
|
||||
}
|
||||
prev = in;
|
||||
block += 16;
|
||||
}
|
||||
#endif
|
||||
return scan_string_finish(p, block, e, plain);
|
||||
}
|
||||
#endif
|
||||
|
||||
} // namespace view
|
||||
} // namespace detail
|
||||
NLOHMANN_JSON_NAMESPACE_END
|
||||
@@ -25,6 +25,7 @@
|
||||
#define INCLUDE_NLOHMANN_JSON_VIEW_HPP_
|
||||
|
||||
#include <cstddef> // size_t
|
||||
#include <cstdint> // uint32_t
|
||||
#include <cstring> // memcpy, strlen
|
||||
#include <iterator> // distance, input_iterator_tag, iterator_traits
|
||||
#include <map> // map
|
||||
@@ -56,6 +57,7 @@
|
||||
#include <nlohmann/detail/view/macro_scope.hpp>
|
||||
#include <nlohmann/detail/view/materialize.hpp>
|
||||
#include <nlohmann/detail/view/node.hpp>
|
||||
#include <nlohmann/detail/view/object_index.hpp>
|
||||
#include <nlohmann/detail/view/pointer.hpp>
|
||||
#include <nlohmann/detail/view/serializer.hpp>
|
||||
#include <nlohmann/detail/view/string_ref.hpp>
|
||||
@@ -925,7 +927,9 @@ class basic_json_document
|
||||
}
|
||||
return sizeof(document_data) + (m_data->inline_cap * sizeof(detail::view::node))
|
||||
+ (m_data->tape != m_data->inline_tape ? m_data->tape_cap * sizeof(detail::view::node) : 0)
|
||||
+ m_data->arena.capacity() + m_data->owned.capacity();
|
||||
+ m_data->arena.capacity() + m_data->owned.capacity()
|
||||
+ (m_data->indexes.capacity() * sizeof(document_data::object_index)) + (m_data->index_slots.capacity() * sizeof(std::uint32_t))
|
||||
+ (m_data->large_objects.capacity() * sizeof(std::uint32_t));
|
||||
}
|
||||
|
||||
/// release unused capacity of the index and the decoded strings; like
|
||||
@@ -995,6 +999,9 @@ class basic_json_document
|
||||
d.size = size;
|
||||
d.tape_size = 0;
|
||||
d.arena.clear();
|
||||
d.indexes.clear();
|
||||
d.index_slots.clear();
|
||||
d.large_objects.clear();
|
||||
d.discarded = true;
|
||||
detail::view::parse_failure failure;
|
||||
bool ok = false;
|
||||
@@ -1010,6 +1017,7 @@ class basic_json_document
|
||||
{
|
||||
d.base[0] = d.src;
|
||||
d.base[1] = d.arena.data();
|
||||
detail::view::build_object_indexes(d);
|
||||
d.discarded = false;
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -25,6 +25,7 @@
|
||||
#define INCLUDE_NLOHMANN_JSON_VIEW_HPP_
|
||||
|
||||
#include <cstddef> // size_t
|
||||
#include <cstdint> // uint32_t
|
||||
#include <cstring> // memcpy, strlen
|
||||
#include <iterator> // distance, input_iterator_tag, iterator_traits
|
||||
#include <map> // map
|
||||
@@ -81,9 +82,11 @@
|
||||
|
||||
#include <array> // array
|
||||
#include <cstddef> // size_t
|
||||
#include <cstdint> // uint32_t
|
||||
#include <cstring> // memcpy
|
||||
#include <new> // operator new, placement new
|
||||
#include <string> // string
|
||||
#include <vector> // vector
|
||||
|
||||
// #include <nlohmann/json.hpp>
|
||||
// #include <nlohmann/detail/view/macro_scope.hpp>
|
||||
@@ -279,6 +282,17 @@ struct document_data
|
||||
std::size_t inline_cap = 0;
|
||||
std::string arena{}; ///< decoded strings that contained escapes // NOLINT(readability-redundant-member-init)
|
||||
std::string owned{}; ///< owned copy of the input, if any // NOLINT(readability-redundant-member-init)
|
||||
|
||||
// hash indexes of large objects (see object_index.hpp)
|
||||
static constexpr std::uint32_t index_min_members = 128;
|
||||
struct object_index
|
||||
{
|
||||
std::size_t start; ///< first slot in index_slots
|
||||
std::uint32_t mask; ///< slot count - 1 (a power of two minus one)
|
||||
};
|
||||
std::vector<object_index> indexes{}; // NOLINT(readability-redundant-member-init)
|
||||
std::vector<std::uint32_t> index_slots{}; // NOLINT(readability-redundant-member-init)
|
||||
std::vector<std::uint32_t> large_objects{}; ///< positions of the objects to index (noted while parsing) // NOLINT(readability-redundant-member-init)
|
||||
std::array<const char*, 4> base = {{nullptr, nullptr, nullptr, nullptr}}; ///< string bases: source, arena (indexed by flags & node_flags::storage)
|
||||
bool discarded = true;
|
||||
|
||||
@@ -306,7 +320,7 @@ struct document_data
|
||||
}
|
||||
};
|
||||
|
||||
document_data() noexcept = default;
|
||||
document_data() = default;
|
||||
document_data(const document_data&) = delete;
|
||||
document_data(document_data&&) = delete;
|
||||
document_data& operator=(const document_data&) = delete;
|
||||
@@ -395,6 +409,290 @@ NLOHMANN_JSON_NAMESPACE_END
|
||||
// #include <nlohmann/json.hpp>
|
||||
// #include <nlohmann/detail/view/macro_scope.hpp>
|
||||
|
||||
// #include <nlohmann/detail/view/simd.hpp>
|
||||
// __ _____ _____ _____
|
||||
// __| | __| | | | JSON for Modern C++
|
||||
// | | |__ | | | | | | version 3.12.0
|
||||
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
|
||||
//
|
||||
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
|
||||
// SPDX-FileCopyrightText: 2018-2025 The simdjson authors <https://github.com/simdjson/simdjson>
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
|
||||
|
||||
#include <array> // array
|
||||
#include <cstddef> // size_t
|
||||
#include <cstdint> // uint8_t, uint64_t
|
||||
|
||||
// #include <nlohmann/json.hpp>
|
||||
// #include <nlohmann/detail/view/macro_scope.hpp>
|
||||
|
||||
|
||||
// Vector code for long runs of string bytes. NEON (AArch64) and SSE2 (x86-64)
|
||||
// belong to the baseline instruction sets and are used by default. The vector
|
||||
// UTF-8 check needs NEON, or SSSE3 if JSON_VIEW_USE_SSSE3 is defined: SSSE3 is
|
||||
// not part of x86-64, so it must not depend on the flags of a translation unit
|
||||
// (two translation units with different flags would have different
|
||||
// definitions of the same inline functions). JSON_VIEW_NO_SIMD selects the
|
||||
// portable code.
|
||||
#if !defined(JSON_VIEW_NO_SIMD) && defined(__aarch64__) && (defined(__GNUC__) || defined(__clang__)) && NLOHMANN_VIEW_LITTLE_ENDIAN
|
||||
#include <arm_neon.h>
|
||||
#define NLOHMANN_VIEW_NEON 1
|
||||
#else
|
||||
#define NLOHMANN_VIEW_NEON 0
|
||||
#endif
|
||||
#if !defined(JSON_VIEW_NO_SIMD) && !NLOHMANN_VIEW_NEON && (defined(__SSE2__) || defined(_M_X64) || (defined(_M_IX86_FP) && _M_IX86_FP >= 2))
|
||||
#include <emmintrin.h>
|
||||
#define NLOHMANN_VIEW_SSE2 1
|
||||
#else
|
||||
#define NLOHMANN_VIEW_SSE2 0
|
||||
#endif
|
||||
#if NLOHMANN_VIEW_SSE2 && defined(JSON_VIEW_USE_SSSE3)
|
||||
#include <tmmintrin.h>
|
||||
#define NLOHMANN_VIEW_SSSE3 1 // NOLINT(cppcoreguidelines-macro-to-enum,modernize-macro-to-enum)
|
||||
#else
|
||||
#define NLOHMANN_VIEW_SSSE3 0 // NOLINT(cppcoreguidelines-macro-to-enum,modernize-macro-to-enum)
|
||||
#endif
|
||||
#define NLOHMANN_VIEW_VECTOR (NLOHMANN_VIEW_NEON || NLOHMANN_VIEW_SSE2)
|
||||
#define NLOHMANN_VIEW_VECTOR_UTF8 (NLOHMANN_VIEW_NEON || NLOHMANN_VIEW_SSSE3)
|
||||
|
||||
NLOHMANN_JSON_NAMESPACE_BEGIN
|
||||
namespace detail
|
||||
{
|
||||
namespace view
|
||||
{
|
||||
|
||||
#if NLOHMANN_VIEW_VECTOR
|
||||
/*!
|
||||
@brief the first byte of a string run that is a quote, a backslash, a control
|
||||
character, or not ASCII, 16 bytes per step
|
||||
|
||||
Stops at such a byte, or where fewer than 16 bytes are left (the caller tells
|
||||
the two apart). A signed compare with 0x20 finds control characters and
|
||||
non-ASCII bytes at once.
|
||||
*/
|
||||
NLOHMANN_VIEW_ALWAYS_INLINE const unsigned char* vector_plain_run(const unsigned char* p, const unsigned char* e) noexcept
|
||||
{
|
||||
while (e - p >= 16)
|
||||
{
|
||||
#if NLOHMANN_VIEW_NEON
|
||||
const uint8x16_t in = vld1q_u8(p);
|
||||
const uint8x16_t special = vorrq_u8(vorrq_u8(vceqq_u8(in, vdupq_n_u8('"')), vceqq_u8(in, vdupq_n_u8('\\'))),
|
||||
vcltq_s8(vreinterpretq_s8_u8(in), vdupq_n_s8(0x20)));
|
||||
// one nibble per byte (the usual NEON replacement of x86's movemask, see
|
||||
// D. Kutenin, "Porting x86 vector bitmask optimizations to Arm NEON", 2022)
|
||||
const std::uint64_t bits = vget_lane_u64(vreinterpret_u64_u8(vshrn_n_u16(vreinterpretq_u16_u8(special), 4)), 0);
|
||||
if (bits != 0)
|
||||
{
|
||||
return p + (count_trailing_zeros(bits) >> 2u);
|
||||
}
|
||||
#else
|
||||
const __m128i in = _mm_loadu_si128(static_cast<const __m128i*>(static_cast<const void*>(p)));
|
||||
const __m128i special = _mm_or_si128(_mm_or_si128(_mm_cmpeq_epi8(in, _mm_set1_epi8('"')), _mm_cmpeq_epi8(in, _mm_set1_epi8('\\'))),
|
||||
_mm_cmplt_epi8(in, _mm_set1_epi8(0x20)));
|
||||
const auto bits = static_cast<std::uint64_t>(static_cast<unsigned>(_mm_movemask_epi8(special)));
|
||||
if (bits != 0)
|
||||
{
|
||||
return p + count_trailing_zeros(bits);
|
||||
}
|
||||
#endif
|
||||
p += 16;
|
||||
}
|
||||
return p;
|
||||
}
|
||||
#endif
|
||||
|
||||
#if NLOHMANN_VIEW_VECTOR_UTF8
|
||||
/// Tables of the UTF-8 check of J. Keiser and D. Lemire, "Validating UTF-8 In
|
||||
/// Less Than One Instruction Per Byte" (2021), as in simdjson ("lookup4"): each
|
||||
/// maps a nibble (high and low nibble of the previous byte, high nibble of the
|
||||
/// current byte) to the errors it allows; a byte pair is ill-formed if all
|
||||
/// three have an error bit in common.
|
||||
template<typename Dummy = void>
|
||||
struct utf8_lookup4
|
||||
{
|
||||
static constexpr std::uint8_t too_short = 1u << 0u, too_long = 1u << 1u, overlong_3 = 1u << 2u, too_large = 1u << 3u;
|
||||
static constexpr std::uint8_t surrogate = 1u << 4u, overlong_2 = 1u << 5u, too_large_1000 = 1u << 6u, overlong_4 = 1u << 6u;
|
||||
static constexpr std::uint8_t two_conts = 1u << 7u, carry = too_short | too_long | two_conts;
|
||||
static const std::array<std::uint8_t, 16> byte_1_high;
|
||||
static const std::array<std::uint8_t, 16> byte_1_low;
|
||||
static const std::array<std::uint8_t, 16> byte_2_high;
|
||||
};
|
||||
|
||||
template<typename Dummy>
|
||||
const std::array<std::uint8_t, 16> utf8_lookup4<Dummy>::byte_1_high =
|
||||
{
|
||||
{
|
||||
too_long, too_long, too_long, too_long, too_long, too_long, too_long, too_long,
|
||||
two_conts, two_conts, two_conts, two_conts,
|
||||
too_short | overlong_2, too_short, too_short | overlong_3 | surrogate, too_short | too_large | too_large_1000 | overlong_4
|
||||
}
|
||||
};
|
||||
|
||||
template<typename Dummy>
|
||||
const std::array<std::uint8_t, 16> utf8_lookup4<Dummy>::byte_1_low =
|
||||
{
|
||||
{
|
||||
carry | overlong_3 | overlong_2 | overlong_4, carry | overlong_2, carry, carry,
|
||||
carry | too_large, carry | too_large | too_large_1000, carry | too_large | too_large_1000, carry | too_large | too_large_1000,
|
||||
carry | too_large | too_large_1000, carry | too_large | too_large_1000, carry | too_large | too_large_1000, carry | too_large | too_large_1000,
|
||||
carry | too_large | too_large_1000, carry | too_large | too_large_1000 | surrogate, carry | too_large | too_large_1000, carry | too_large | too_large_1000
|
||||
}
|
||||
};
|
||||
|
||||
template<typename Dummy>
|
||||
const std::array<std::uint8_t, 16> utf8_lookup4<Dummy>::byte_2_high =
|
||||
{
|
||||
{
|
||||
too_short, too_short, too_short, too_short, too_short, too_short, too_short, too_short,
|
||||
static_cast<std::uint8_t>(too_long | overlong_2 | two_conts | overlong_3 | too_large_1000 | overlong_4),
|
||||
static_cast<std::uint8_t>(too_long | overlong_2 | two_conts | overlong_3 | too_large),
|
||||
static_cast<std::uint8_t>(too_long | overlong_2 | two_conts | surrogate | too_large),
|
||||
static_cast<std::uint8_t>(too_long | overlong_2 | two_conts | surrogate | too_large),
|
||||
too_short, too_short, too_short, too_short
|
||||
}
|
||||
};
|
||||
|
||||
/// the end of scan_string_vector from block, where the vector loop stopped
|
||||
/// (ill-formed UTF-8, or fewer than 16 bytes left): one byte or sequence at a
|
||||
/// time, from the start of a sequence that crosses into the block
|
||||
inline const unsigned char* scan_string_finish(const unsigned char* p, const unsigned char* block, const unsigned char* e, const std::uint8_t* plain) noexcept
|
||||
{
|
||||
for (int i = 1; i <= 3 && block - i >= p; ++i)
|
||||
{
|
||||
const unsigned char c = block[-i];
|
||||
if (c < 0x80)
|
||||
{
|
||||
break;
|
||||
}
|
||||
if (c >= 0xC0)
|
||||
{
|
||||
const int len = 2 + static_cast<int>(c >= 0xE0) + static_cast<int>(c >= 0xF0);
|
||||
if (len > i)
|
||||
{
|
||||
block -= i;
|
||||
}
|
||||
break;
|
||||
}
|
||||
}
|
||||
for (p = block; p != e;)
|
||||
{
|
||||
if (*p < 0x80)
|
||||
{
|
||||
if (plain[*p] == 0)
|
||||
{
|
||||
return p;
|
||||
}
|
||||
++p;
|
||||
continue;
|
||||
}
|
||||
const std::size_t n = validate_one_utf8(p, static_cast<std::size_t>(e - p));
|
||||
if (n == 0)
|
||||
{
|
||||
return p;
|
||||
}
|
||||
p += n;
|
||||
}
|
||||
return p;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief the rest of a string from p (a character boundary), 16 bytes per step
|
||||
|
||||
The first quote, backslash, or control character is found with vector
|
||||
compares, and the UTF-8 check covers the bytes up to it. Returns where the
|
||||
string scan stops, like scan_string_run: before ill-formed UTF-8 and for the
|
||||
last bytes of the input, the bytes are checked one sequence at a time. Out of
|
||||
line, so that no constants of the check occupy registers in the parse loop.
|
||||
*/
|
||||
NLOHMANN_VIEW_NOINLINE inline const unsigned char* scan_string_vector(const unsigned char* p, const unsigned char* e, const std::uint8_t* plain) noexcept
|
||||
{
|
||||
using lookup = utf8_lookup4<>;
|
||||
const unsigned char* block = p;
|
||||
#if NLOHMANN_VIEW_NEON
|
||||
const uint8x16_t t1h = vld1q_u8(lookup::byte_1_high.data());
|
||||
const uint8x16_t t1l = vld1q_u8(lookup::byte_1_low.data());
|
||||
const uint8x16_t t2h = vld1q_u8(lookup::byte_2_high.data());
|
||||
uint8x16_t prev = vdupq_n_u8(0);
|
||||
while (e - block >= 16)
|
||||
{
|
||||
const uint8x16_t in = vld1q_u8(block);
|
||||
const uint8x16_t special = vorrq_u8(vorrq_u8(vceqq_u8(in, vdupq_n_u8('"')), vceqq_u8(in, vdupq_n_u8('\\'))), vcltq_u8(in, vdupq_n_u8(0x20)));
|
||||
const uint8x16_t prev1 = vextq_u8(prev, in, 15);
|
||||
const uint8x16_t sc = vandq_u8(vandq_u8(vqtbl1q_u8(t1h, vshrq_n_u8(prev1, 4)), vqtbl1q_u8(t1l, vandq_u8(prev1, vdupq_n_u8(0x0F)))), vqtbl1q_u8(t2h, vshrq_n_u8(in, 4)));
|
||||
const uint8x16_t must23 = vorrq_u8(vqsubq_u8(vextq_u8(prev, in, 14), vdupq_n_u8(0xE0 - 0x80)), vqsubq_u8(vextq_u8(prev, in, 13), vdupq_n_u8(0xF0 - 0x80)));
|
||||
const uint8x16_t err = veorq_u8(vandq_u8(must23, vdupq_n_u8(0x80)), sc);
|
||||
const std::uint64_t special_bits = vget_lane_u64(vreinterpret_u64_u8(vshrn_n_u16(vreinterpretq_u16_u8(special), 4)), 0);
|
||||
const std::uint64_t err_bits = vget_lane_u64(vreinterpret_u64_u8(vshrn_n_u16(vreinterpretq_u16_u8(vtstq_u8(err, err)), 4)), 0);
|
||||
if (special_bits != 0)
|
||||
{
|
||||
// errors up to the special byte count (an incomplete sequence
|
||||
// before a quote shows at the quote); the bytes after it do not
|
||||
const unsigned k = static_cast<unsigned>(count_trailing_zeros(special_bits)) >> 2u;
|
||||
const std::uint64_t upto = k == 15 ? ~std::uint64_t{0} :
|
||||
(std::uint64_t{1} << (4u * (k + 1u))) - 1u;
|
||||
if ((err_bits & upto) == 0)
|
||||
{
|
||||
return block + k;
|
||||
}
|
||||
break;
|
||||
}
|
||||
if (err_bits != 0)
|
||||
{
|
||||
break;
|
||||
}
|
||||
prev = in;
|
||||
block += 16;
|
||||
}
|
||||
#else
|
||||
// the same with SSSE3 (pshufb for the table lookups; nibbles from 16-bit
|
||||
// shifts, as there are no byte shifts)
|
||||
const __m128i t1h = _mm_loadu_si128(static_cast<const __m128i*>(static_cast<const void*>(lookup::byte_1_high.data())));
|
||||
const __m128i t1l = _mm_loadu_si128(static_cast<const __m128i*>(static_cast<const void*>(lookup::byte_1_low.data())));
|
||||
const __m128i t2h = _mm_loadu_si128(static_cast<const __m128i*>(static_cast<const void*>(lookup::byte_2_high.data())));
|
||||
const __m128i nibble = _mm_set1_epi8(0x0F);
|
||||
const __m128i zero = _mm_setzero_si128();
|
||||
__m128i prev = zero;
|
||||
while (e - block >= 16)
|
||||
{
|
||||
const __m128i in = _mm_loadu_si128(static_cast<const __m128i*>(static_cast<const void*>(block)));
|
||||
const __m128i special = _mm_or_si128(_mm_or_si128(_mm_cmpeq_epi8(in, _mm_set1_epi8('"')), _mm_cmpeq_epi8(in, _mm_set1_epi8('\\'))),
|
||||
_mm_cmpeq_epi8(_mm_subs_epu8(in, _mm_set1_epi8(0x1F)), zero)); // in < 0x20
|
||||
const __m128i prev1 = _mm_alignr_epi8(in, prev, 15);
|
||||
const __m128i sc = _mm_and_si128(_mm_and_si128(_mm_shuffle_epi8(t1h, _mm_and_si128(_mm_srli_epi16(prev1, 4), nibble)),
|
||||
_mm_shuffle_epi8(t1l, _mm_and_si128(prev1, nibble))),
|
||||
_mm_shuffle_epi8(t2h, _mm_and_si128(_mm_srli_epi16(in, 4), nibble)));
|
||||
const __m128i must23 = _mm_or_si128(_mm_subs_epu8(_mm_alignr_epi8(in, prev, 14), _mm_set1_epi8(0xE0 - 0x80)),
|
||||
_mm_subs_epu8(_mm_alignr_epi8(in, prev, 13), _mm_set1_epi8(0xF0 - 0x80)));
|
||||
const __m128i err = _mm_xor_si128(_mm_and_si128(must23, _mm_set1_epi8(static_cast<char>(-128))), sc);
|
||||
const auto special_bits = static_cast<unsigned>(_mm_movemask_epi8(special));
|
||||
const auto err_bits = ~static_cast<unsigned>(_mm_movemask_epi8(_mm_cmpeq_epi8(err, zero))) & 0xFFFFu;
|
||||
if (special_bits != 0)
|
||||
{
|
||||
const unsigned k = static_cast<unsigned>(count_trailing_zeros(static_cast<std::uint64_t>(special_bits)));
|
||||
if ((err_bits & ((2u << k) - 1u)) == 0)
|
||||
{
|
||||
return block + k;
|
||||
}
|
||||
break;
|
||||
}
|
||||
if (err_bits != 0)
|
||||
{
|
||||
break;
|
||||
}
|
||||
prev = in;
|
||||
block += 16;
|
||||
}
|
||||
#endif
|
||||
return scan_string_finish(p, block, e, plain);
|
||||
}
|
||||
#endif
|
||||
|
||||
} // namespace view
|
||||
} // namespace detail
|
||||
NLOHMANN_JSON_NAMESPACE_END
|
||||
|
||||
|
||||
// Scanning primitives of the view's parser. The unrolled checks at fixed
|
||||
// offsets follow yyjson (https://github.com/ibireme/yyjson, MIT license): the
|
||||
@@ -441,9 +739,12 @@ NLOHMANN_VIEW_ALWAYS_INLINE std::uint16_t load16(const unsigned char* p) noexcep
|
||||
|
||||
/// Advance over plain string bytes and well-formed UTF-8. Stops at a quote,
|
||||
/// a backslash, a control character, ill-formed UTF-8, or the end. The first
|
||||
/// 16 bytes are checked one by one, so that the position advances by
|
||||
/// constants in predicted branches (most strings are short); longer runs
|
||||
/// continue eight bytes at a time.
|
||||
/// bytes are checked one by one, so that the position advances by constants
|
||||
/// in predicted branches: 16 for keys, whose lengths repeat from record to
|
||||
/// record, and 8 for string values (Value) where a vector loop follows, as
|
||||
/// their lengths vary more. Longer runs continue 16 bytes at a time with NEON
|
||||
/// or SSE2, else eight bytes at a time.
|
||||
template<bool Value = false>
|
||||
NLOHMANN_VIEW_ALWAYS_INLINE const unsigned char* scan_string_run(const unsigned char* p, const unsigned char* e) noexcept
|
||||
{
|
||||
const std::uint8_t* plain = string_plain();
|
||||
@@ -452,9 +753,23 @@ NLOHMANN_VIEW_ALWAYS_INLINE const unsigned char* scan_string_run(const unsigned
|
||||
if (e - p >= 16)
|
||||
{
|
||||
#define NLOHMANN_VIEW_STEP(i) if (NLOHMANN_VIEW_LIKELY(plain[p[i]] != 0)) {} else { p += (i); goto stop; }
|
||||
NLOHMANN_VIEW_REPEAT16(NLOHMANN_VIEW_STEP)
|
||||
NLOHMANN_VIEW_STEP(0) NLOHMANN_VIEW_STEP(1) NLOHMANN_VIEW_STEP(2) NLOHMANN_VIEW_STEP(3)
|
||||
NLOHMANN_VIEW_STEP(4) NLOHMANN_VIEW_STEP(5) NLOHMANN_VIEW_STEP(6) NLOHMANN_VIEW_STEP(7)
|
||||
if (!Value || !NLOHMANN_VIEW_VECTOR)
|
||||
{
|
||||
NLOHMANN_VIEW_STEP(8) NLOHMANN_VIEW_STEP(9) NLOHMANN_VIEW_STEP(10) NLOHMANN_VIEW_STEP(11)
|
||||
NLOHMANN_VIEW_STEP(12) NLOHMANN_VIEW_STEP(13) NLOHMANN_VIEW_STEP(14) NLOHMANN_VIEW_STEP(15)
|
||||
p += 8;
|
||||
}
|
||||
#undef NLOHMANN_VIEW_STEP
|
||||
p += 16;
|
||||
p += 8;
|
||||
#if NLOHMANN_VIEW_VECTOR
|
||||
p = vector_plain_run(p, e);
|
||||
if (p != e && plain[*p] == 0)
|
||||
{
|
||||
goto stop;
|
||||
}
|
||||
#else
|
||||
while (e - p >= 8)
|
||||
{
|
||||
const std::uint64_t special = swar_string_special(read_eight_bytes(p));
|
||||
@@ -465,6 +780,7 @@ NLOHMANN_VIEW_ALWAYS_INLINE const unsigned char* scan_string_run(const unsigned
|
||||
}
|
||||
p += 8;
|
||||
}
|
||||
#endif
|
||||
continue;
|
||||
}
|
||||
while (p != e && plain[*p] != 0)
|
||||
@@ -480,6 +796,10 @@ stop:
|
||||
{
|
||||
return p; // quote, backslash, or control character
|
||||
}
|
||||
#if NLOHMANN_VIEW_VECTOR_UTF8
|
||||
// non-ASCII: the vector check, out of line
|
||||
return scan_string_vector(p, e, plain);
|
||||
#else
|
||||
// non-ASCII: a run of well-formed sequences (the library's check, so
|
||||
// that exactly what json::parse accepts is accepted)
|
||||
do
|
||||
@@ -492,6 +812,7 @@ stop:
|
||||
p += n;
|
||||
}
|
||||
while (p != e && *p >= 0x80);
|
||||
#endif
|
||||
}
|
||||
}
|
||||
|
||||
@@ -659,6 +980,13 @@ class builder
|
||||
frame shallow[64]; // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays): not initialized on purpose; filled as containers open
|
||||
std::vector<frame> deep{};
|
||||
|
||||
/// remember an object to index after parsing (out of line, so that the
|
||||
/// parse loop only has a call for it)
|
||||
NLOHMANN_VIEW_NOINLINE void note_large_object(std::uint32_t idx)
|
||||
{
|
||||
doc.large_objects.push_back(idx);
|
||||
}
|
||||
|
||||
NLOHMANN_VIEW_NOINLINE bool fail(error_code c, const unsigned char* at) noexcept
|
||||
{
|
||||
m_failure.code = c;
|
||||
@@ -1045,7 +1373,7 @@ class builder
|
||||
switch (cur()) \
|
||||
{ \
|
||||
case '"': \
|
||||
if (NLOHMANN_VIEW_UNLIKELY(!string())) { return false; } \
|
||||
if (NLOHMANN_VIEW_UNLIKELY(!string<true>())) { return false; } \
|
||||
goto NEXT; \
|
||||
case '{': \
|
||||
open(value_t::object); \
|
||||
@@ -1129,7 +1457,7 @@ obj_key:
|
||||
{
|
||||
return fail(error_code::expected_key);
|
||||
}
|
||||
if (NLOHMANN_VIEW_UNLIKELY(!string()))
|
||||
if (NLOHMANN_VIEW_UNLIKELY(!string<false>()))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
@@ -1170,19 +1498,27 @@ obj_next:
|
||||
if (enabled(TrailingCommas) && cur() == '}')
|
||||
{
|
||||
++p;
|
||||
goto close_container;
|
||||
goto close_object;
|
||||
}
|
||||
goto obj_key;
|
||||
}
|
||||
if (cur() == '}')
|
||||
{
|
||||
++p;
|
||||
goto close_container;
|
||||
goto close_object;
|
||||
}
|
||||
return fail(error_code::expected_object_end);
|
||||
|
||||
#undef NLOHMANN_VIEW_VALUE
|
||||
|
||||
close_object:
|
||||
// a large object gets a hash index (objects only, so that closing
|
||||
// an array pays nothing for this)
|
||||
if (NLOHMANN_VIEW_UNLIKELY(cur_count >= document_data::index_min_members))
|
||||
{
|
||||
cold.note_large_object(cur_idx);
|
||||
}
|
||||
|
||||
close_container:
|
||||
close();
|
||||
if (NLOHMANN_VIEW_UNLIKELY(depth == 0))
|
||||
@@ -1232,7 +1568,7 @@ root_done:
|
||||
switch (cur())
|
||||
{
|
||||
case '"':
|
||||
return string();
|
||||
return string<true>();
|
||||
case 't':
|
||||
return literal("true", 4, value_t::boolean, node_flags::is_true);
|
||||
case 'f':
|
||||
@@ -1518,12 +1854,13 @@ indent_done:
|
||||
return true;
|
||||
}
|
||||
|
||||
/// a string (value or key) at p
|
||||
/// a string at p: a value (Value) or a key
|
||||
template<bool Value>
|
||||
NLOHMANN_VIEW_ALWAYS_INLINE bool string()
|
||||
{
|
||||
++p; // opening quote
|
||||
const unsigned char* const s = p;
|
||||
p = scan_string_run(p, e);
|
||||
p = scan_string_run<Value>(p, e);
|
||||
if (NLOHMANN_VIEW_LIKELY(p != e && *p == '"'))
|
||||
{
|
||||
emit(value_t::string, 0, 0, static_cast<std::size_t>(s - b), static_cast<std::uint64_t>(p - s));
|
||||
@@ -2375,6 +2712,142 @@ NLOHMANN_JSON_NAMESPACE_END
|
||||
|
||||
// #include <nlohmann/detail/view/node.hpp>
|
||||
|
||||
// #include <nlohmann/detail/view/object_index.hpp>
|
||||
// __ _____ _____ _____
|
||||
// __| | __| | | | JSON for Modern C++
|
||||
// | | |__ | | | | | | version 3.12.0
|
||||
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
|
||||
//
|
||||
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
|
||||
|
||||
#include <cstddef> // size_t
|
||||
#include <cstdint> // uint32_t, uint64_t
|
||||
#include <cstring> // memcmp
|
||||
|
||||
// #include <nlohmann/json.hpp>
|
||||
// #include <nlohmann/detail/view/document_data.hpp>
|
||||
|
||||
// #include <nlohmann/detail/view/macro_scope.hpp>
|
||||
|
||||
// #include <nlohmann/detail/view/node.hpp>
|
||||
|
||||
|
||||
// Hash indexes of large objects, so that a lookup does not compare thousands
|
||||
// of keys (as Boost.JSON switches from a linear search to a hash table for
|
||||
// large objects). An object with document_data::index_min_members members or
|
||||
// more gets an open-addressing table after parsing; its node stores the
|
||||
// number of the table (1-based) in `extra`. A slot holds the offset of a key
|
||||
// node from its object node (0: empty). Of duplicate keys, the first is kept,
|
||||
// as for the linear search.
|
||||
|
||||
NLOHMANN_JSON_NAMESPACE_BEGIN
|
||||
namespace detail
|
||||
{
|
||||
namespace view
|
||||
{
|
||||
|
||||
/// hash of a key: its bytes, eight at a time, in a fixed byte order
|
||||
inline std::uint64_t key_hash(const char* s, std::size_t n) noexcept
|
||||
{
|
||||
std::uint64_t h = 0x9E3779B97F4A7C15u * (n + 1);
|
||||
const auto* p = reinterpret_cast<const unsigned char*>(s); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
|
||||
while (n >= 8)
|
||||
{
|
||||
h = (h ^ read_eight_bytes(p)) * 0xBF58476D1CE4E5B9u;
|
||||
h ^= h >> 29u;
|
||||
p += 8;
|
||||
n -= 8;
|
||||
}
|
||||
std::uint64_t w = 0;
|
||||
for (std::size_t i = 0; i < n; ++i)
|
||||
{
|
||||
w |= static_cast<std::uint64_t>(p[i]) << (8u * i);
|
||||
}
|
||||
h = (h ^ w) * 0x94D049BB133111EBu;
|
||||
return h ^ (h >> 31u);
|
||||
}
|
||||
|
||||
/// build the table of a large object
|
||||
inline void build_object_index(document_data& d, node* obj)
|
||||
{
|
||||
if (d.indexes.size() >= 0xFFFFu)
|
||||
{
|
||||
return; // LCOV_EXCL_LINE (the number must fit `extra`; more large objects are searched linearly)
|
||||
}
|
||||
std::size_t cap = 16;
|
||||
while (cap < 2 * static_cast<std::size_t>(obj->len))
|
||||
{
|
||||
cap *= 2;
|
||||
}
|
||||
const std::size_t start = d.index_slots.size();
|
||||
d.index_slots.resize(start + cap, 0);
|
||||
std::uint32_t* const slots = d.index_slots.data() + start;
|
||||
const std::size_t mask = cap - 1;
|
||||
for (const node* k = document_data::first_child(obj), *end = document_data::child_end(obj); k != end; k = document_data::after(k + 1))
|
||||
{
|
||||
const char* const key = d.str(*k);
|
||||
const std::uint64_t hash = key_hash(key, k->len); // (a cast of the call would be useless where std::uint64_t is std::size_t)
|
||||
std::size_t i = static_cast<std::size_t>(hash) & mask;
|
||||
bool duplicate = false;
|
||||
while (slots[i] != 0)
|
||||
{
|
||||
const node* const other = obj + slots[i];
|
||||
if (other->len == k->len && (k->len == 0 || std::memcmp(d.str(*other), key, k->len) == 0))
|
||||
{
|
||||
duplicate = true; // keep the first
|
||||
break;
|
||||
}
|
||||
i = (i + 1) & mask;
|
||||
}
|
||||
if (!duplicate)
|
||||
{
|
||||
slots[i] = static_cast<std::uint32_t>(k - obj);
|
||||
}
|
||||
}
|
||||
d.indexes.push_back(document_data::object_index{start, static_cast<std::uint32_t>(mask)});
|
||||
obj->extra = static_cast<std::uint16_t>(d.indexes.size());
|
||||
}
|
||||
|
||||
/// build the tables of the large objects the parser noted
|
||||
inline void build_object_indexes(document_data& d)
|
||||
{
|
||||
for (const std::uint32_t i : d.large_objects)
|
||||
{
|
||||
build_object_index(d, d.tape + i);
|
||||
}
|
||||
}
|
||||
|
||||
/// the key node of the first member with this key of an indexed object, or
|
||||
/// nullptr
|
||||
inline const node* find_indexed(const document_data& d, const node* obj, const char* key, std::size_t n) noexcept
|
||||
{
|
||||
const document_data::object_index& ix = d.indexes[obj->extra - 1u];
|
||||
const std::uint32_t* const slots = d.index_slots.data() + ix.start;
|
||||
const std::uint64_t hash = key_hash(key, n); // (a cast of the call would be useless where std::uint64_t is std::size_t)
|
||||
std::size_t i = static_cast<std::size_t>(hash) & ix.mask;
|
||||
for (;;)
|
||||
{
|
||||
const std::uint32_t s = slots[i];
|
||||
if (s == 0)
|
||||
{
|
||||
return nullptr;
|
||||
}
|
||||
const node* const k = obj + s;
|
||||
if (k->len == n && (n == 0 || std::memcmp(d.str(*k), key, n) == 0))
|
||||
{
|
||||
return k;
|
||||
}
|
||||
i = (i + 1) & ix.mask;
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace view
|
||||
} // namespace detail
|
||||
NLOHMANN_JSON_NAMESPACE_END
|
||||
|
||||
|
||||
NLOHMANN_JSON_NAMESPACE_BEGIN
|
||||
namespace detail
|
||||
@@ -2444,6 +2917,10 @@ class short_key
|
||||
/// nullptr; most keys are rejected by their length, from the index alone
|
||||
inline const node* find_member(const document_data& d, const node* object, const char* key, std::size_t n) noexcept
|
||||
{
|
||||
if (NLOHMANN_VIEW_UNLIKELY(object->extra != 0))
|
||||
{
|
||||
return find_indexed(d, object, key, n); // a large object
|
||||
}
|
||||
const node* const end = document_data::child_end(object);
|
||||
const auto* const k = reinterpret_cast<const unsigned char*>(key); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
|
||||
if (NLOHMANN_VIEW_LIKELY(n <= 16))
|
||||
@@ -2812,6 +3289,8 @@ NLOHMANN_JSON_NAMESPACE_END
|
||||
|
||||
// #include <nlohmann/detail/view/node.hpp>
|
||||
|
||||
// #include <nlohmann/detail/view/object_index.hpp>
|
||||
|
||||
// #include <nlohmann/detail/view/pointer.hpp>
|
||||
// __ _____ _____ _____
|
||||
// __| | __| | | | JSON for Modern C++
|
||||
@@ -4506,7 +4985,9 @@ class basic_json_document
|
||||
}
|
||||
return sizeof(document_data) + (m_data->inline_cap * sizeof(detail::view::node))
|
||||
+ (m_data->tape != m_data->inline_tape ? m_data->tape_cap * sizeof(detail::view::node) : 0)
|
||||
+ m_data->arena.capacity() + m_data->owned.capacity();
|
||||
+ m_data->arena.capacity() + m_data->owned.capacity()
|
||||
+ (m_data->indexes.capacity() * sizeof(document_data::object_index)) + (m_data->index_slots.capacity() * sizeof(std::uint32_t))
|
||||
+ (m_data->large_objects.capacity() * sizeof(std::uint32_t));
|
||||
}
|
||||
|
||||
/// release unused capacity of the index and the decoded strings; like
|
||||
@@ -4576,6 +5057,9 @@ class basic_json_document
|
||||
d.size = size;
|
||||
d.tape_size = 0;
|
||||
d.arena.clear();
|
||||
d.indexes.clear();
|
||||
d.index_slots.clear();
|
||||
d.large_objects.clear();
|
||||
d.discarded = true;
|
||||
detail::view::parse_failure failure;
|
||||
bool ok = false;
|
||||
@@ -4591,6 +5075,7 @@ class basic_json_document
|
||||
{
|
||||
d.base[0] = d.src;
|
||||
d.base[1] = d.arena.data();
|
||||
detail::view::build_object_indexes(d);
|
||||
d.discarded = false;
|
||||
return;
|
||||
}
|
||||
@@ -4756,6 +5241,11 @@ class tuple_element<N, ::nlohmann::detail::view::view_item<View>> // NOLINT(cert
|
||||
#undef NLOHMANN_VIEW_THROW
|
||||
#undef NLOHMANN_VIEW_LITTLE_ENDIAN
|
||||
#undef NLOHMANN_VIEW_REPEAT16
|
||||
#undef NLOHMANN_VIEW_NEON
|
||||
#undef NLOHMANN_VIEW_SSE2
|
||||
#undef NLOHMANN_VIEW_SSSE3
|
||||
#undef NLOHMANN_VIEW_VECTOR
|
||||
#undef NLOHMANN_VIEW_VECTOR_UTF8
|
||||
|
||||
|
||||
#endif // INCLUDE_NLOHMANN_JSON_VIEW_HPP_
|
||||
|
||||
@@ -324,6 +324,26 @@ json_test_add_test_for(src/unit-diagnostic-positions.cpp
|
||||
MAIN test_main CXX_STANDARDS ${test_cxx_standards} ${test_force}
|
||||
)
|
||||
|
||||
# the json_view parser again with the portable string scanning instead of
|
||||
# NEON/SSE2, and on x86-64 with the SSSE3 UTF-8 check (JSON_VIEW_USE_SSSE3)
|
||||
json_test_set_test_options(test-json_view_builder_portable
|
||||
COMPILE_DEFINITIONS JSON_VIEW_NO_SIMD
|
||||
)
|
||||
json_test_add_test_for(src/unit-json_view_builder.cpp
|
||||
NAME test-json_view_builder_portable
|
||||
MAIN test_main CXX_STANDARDS ${test_cxx_standards} ${test_force}
|
||||
)
|
||||
if(CMAKE_SYSTEM_PROCESSOR MATCHES "^(x86_64|AMD64|amd64)$" AND NOT MSVC)
|
||||
json_test_set_test_options(test-json_view_builder_ssse3
|
||||
COMPILE_DEFINITIONS JSON_VIEW_USE_SSSE3
|
||||
COMPILE_OPTIONS -mssse3
|
||||
)
|
||||
json_test_add_test_for(src/unit-json_view_builder.cpp
|
||||
NAME test-json_view_builder_ssse3
|
||||
MAIN test_main CXX_STANDARDS ${test_cxx_standards} ${test_force}
|
||||
)
|
||||
endif()
|
||||
|
||||
# *DO NOT* use json_test_set_test_options() below this line
|
||||
|
||||
#############################################################################
|
||||
|
||||
@@ -0,0 +1 @@
|
||||
build/
|
||||
@@ -0,0 +1,73 @@
|
||||
# json_view compared with other libraries
|
||||
|
||||
The in-tree benchmarks in [`tests/benchmarks`](../README.md) measure `json_document` against `json::parse` only. The
|
||||
programs here compare it with [yyjson](https://github.com/ibireme/yyjson),
|
||||
[simdjson](https://github.com/simdjson/simdjson), and [Boost.JSON](https://github.com/boostorg/json): the question
|
||||
users ask when they pick a library. They are not built by CMake or run by CI.
|
||||
|
||||
## Reproducing the numbers
|
||||
|
||||
`compare.py` builds both programs against `include/` of this checkout, runs them, and writes the results together with
|
||||
everything needed to reproduce them to `results/<date>-<host>.md` (and `.csv`): the date, the commit, the CPU, the
|
||||
OS, the compiler, the flags, and the versions of all libraries.
|
||||
|
||||
```sh
|
||||
python3 tests/benchmarks/json_view/compare.py --data <json_test_data directory> [--native] [--rounds 30]
|
||||
```
|
||||
|
||||
- `--data` is the downloaded [test data](https://github.com/nlohmann/json_test_data), e.g. the `test_files` directory
|
||||
of a CMake build directory. It needs `nativejson-benchmark/{twitter,citm_catalog,canada}.json` and
|
||||
`jeopardy/jeopardy.json`.
|
||||
- The other libraries come from the system: pkg-config, or Homebrew (`brew install yyjson simdjson boost`). With
|
||||
`--download`, pinned releases are downloaded instead and checked against their SHA-256. Without Boost headers (or
|
||||
with `--no-boost`), the Boost.JSON columns are skipped, and the results say so.
|
||||
- `--corpus file...` adds files to the corpus benchmark, e.g. those of
|
||||
[simdjson-data](https://github.com/simdjson/simdjson-data) or the
|
||||
[yyjson benchmark](https://github.com/ibireme/yyjson_benchmark).
|
||||
- Only the Python 3 standard library is used; a C++17 compiler is needed (`CXX` and `CC` are honored).
|
||||
|
||||
For numbers worth publishing, use a quiet machine (see [Getting stable numbers](../README.md#getting-stable-numbers)),
|
||||
the default 30 rounds or more, and `--native` only if the other libraries were built for the same CPU.
|
||||
|
||||
### On GitHub-hosted runners
|
||||
|
||||
The workflow [json_view benchmarks](../../../.github/workflows/json_view_benchmarks.yml) runs `compare.py --download`
|
||||
on demand: by hand (Actions → "json_view benchmarks" → "Run workflow"), on an x86-64 or AArch64 Ubuntu runner with GCC
|
||||
or Clang, or when a pull request gets the label `benchmark`, on both architectures with GCC. The results appear as the
|
||||
job summary and as an artifact. Shared runners are noisy, so these numbers show
|
||||
where `json_view` stands on another architecture; they are not meant for publication.
|
||||
|
||||
## What is measured
|
||||
|
||||
`bench_view.cpp` runs four workloads on twitter, citm_catalog, canada, jeopardy, a single tweet (`status`), and a
|
||||
JSON-RPC request (`rpc`):
|
||||
|
||||
| workload | what it does |
|
||||
|---|---|
|
||||
| parse | build and free a document |
|
||||
| traverse | parse, then visit every value, convert every number, touch every string and key |
|
||||
| select | parse, then read a few fields per record (e.g. id, user name, and retweet count of each tweet) |
|
||||
| dump | serialize a parsed document (compact) |
|
||||
|
||||
`bench_corpus.cpp` runs parse, traverse, and dump on any list of files, so that no library is tuned to a handful of
|
||||
documents.
|
||||
|
||||
Before anything is timed, all engines must accept each document and agree on the traversal: the number of values, the
|
||||
bytes of all strings and keys, and the sum of all numbers. All engines run interleaved in every round, and the best
|
||||
round is reported, as time and as a factor of the `json_view` time (below 1 means faster than `json_view`).
|
||||
|
||||
The engines do not all offer the same features, which the numbers should be read with:
|
||||
|
||||
| engine | document | random access | editable | notes |
|
||||
|---|---|---|---|---|
|
||||
| `json_view` | immutable index into the text | yes | no | a fresh document per parse; "reused" parses into the same document |
|
||||
| yyjson | immutable (`yyjson_read`) | yes | via a mutable copy | |
|
||||
| simdjson DOM | immutable, parser reused | yes | no | |
|
||||
| simdjson On-Demand | none: forward-only, lazy | no | no | only traverse and select |
|
||||
| Boost.JSON | owning, mutable DOM | yes | yes | monotonic resource |
|
||||
| `json::parse` | owning, mutable DOM | yes | yes | |
|
||||
|
||||
## Published results
|
||||
|
||||
Results are only published with the file `compare.py` wrote, which names the machine and the versions; see
|
||||
`results/`. Numbers from one machine and compiler do not carry over to another: rerun the script.
|
||||
@@ -0,0 +1,326 @@
|
||||
// __ _____ _____ _____
|
||||
// __| | __| | | | JSON for Modern C++ (supporting code)
|
||||
// | | |__ | | | | | | version 3.12.0
|
||||
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
|
||||
//
|
||||
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
// Corpus benchmark: the read-only workloads of bench_view.cpp on any list of
|
||||
// JSON files (for example the benchmark sets of simdjson and yyjson).
|
||||
//
|
||||
// ./bench_corpus [--rounds N] file...
|
||||
//
|
||||
// For every file, all engines must accept it and agree on a traversal (value
|
||||
// count, string bytes, sum of numbers) before anything is timed. Workloads:
|
||||
// parse (build and free a document), traverse (visit every value, convert
|
||||
// every number), dump (compact), and for json_view also dump with the source
|
||||
// number text. Results go to bench_corpus.csv.
|
||||
#include <nlohmann/json_view.hpp>
|
||||
|
||||
#if JSON_VIEW_BENCH_BOOST
|
||||
#include <boost/json.hpp>
|
||||
#include <boost/json/src.hpp>
|
||||
#endif
|
||||
#include <simdjson.h>
|
||||
#include <yyjson.h>
|
||||
|
||||
#include <algorithm>
|
||||
#include <chrono>
|
||||
#include <cmath>
|
||||
#include <cstdio>
|
||||
#include <cstring>
|
||||
#include <fstream>
|
||||
#include <functional>
|
||||
#include <sstream>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
using nlohmann::json;
|
||||
using nlohmann::json_document;
|
||||
using nlohmann::json_view;
|
||||
|
||||
static volatile double g_sink;
|
||||
|
||||
struct stats
|
||||
{
|
||||
double num = 0;
|
||||
std::size_t str = 0, nodes = 0;
|
||||
};
|
||||
|
||||
static void walk(json_view v, stats& st)
|
||||
{
|
||||
++st.nodes;
|
||||
switch (v.type())
|
||||
{
|
||||
case json::value_t::object:
|
||||
for (auto it = v.begin(); it != v.end(); ++it)
|
||||
{
|
||||
st.str += it.key().size();
|
||||
walk(*it, st);
|
||||
}
|
||||
break;
|
||||
case json::value_t::array:
|
||||
for (const json_view e : v)
|
||||
{
|
||||
walk(e, st);
|
||||
}
|
||||
break;
|
||||
case json::value_t::string:
|
||||
st.str += v.get_string().size();
|
||||
break;
|
||||
case json::value_t::number_integer:
|
||||
st.num += static_cast<double>(v.get<std::int64_t>());
|
||||
break;
|
||||
case json::value_t::number_unsigned:
|
||||
st.num += static_cast<double>(v.get<std::uint64_t>());
|
||||
break;
|
||||
case json::value_t::number_float:
|
||||
st.num += v.get<double>();
|
||||
break;
|
||||
default:
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
static void walk(yyjson_val* v, stats& st)
|
||||
{
|
||||
++st.nodes;
|
||||
switch (yyjson_get_type(v))
|
||||
{
|
||||
case YYJSON_TYPE_OBJ:
|
||||
{
|
||||
std::size_t idx, max;
|
||||
yyjson_val* k, * val;
|
||||
yyjson_obj_foreach(v, idx, max, k, val)
|
||||
{
|
||||
st.str += yyjson_get_len(k);
|
||||
walk(val, st);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case YYJSON_TYPE_ARR:
|
||||
{
|
||||
std::size_t idx, max;
|
||||
yyjson_val* val;
|
||||
yyjson_arr_foreach(v, idx, max, val)
|
||||
{
|
||||
walk(val, st);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case YYJSON_TYPE_STR:
|
||||
st.str += yyjson_get_len(v);
|
||||
break;
|
||||
case YYJSON_TYPE_NUM:
|
||||
st.num += yyjson_is_sint(v) ? static_cast<double>(yyjson_get_sint(v)) : yyjson_is_uint(v) ? static_cast<double>(yyjson_get_uint(v)) : yyjson_get_real(v);
|
||||
break;
|
||||
default:
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
static void walk(simdjson::dom::element e, stats& st)
|
||||
{
|
||||
++st.nodes;
|
||||
switch (e.type())
|
||||
{
|
||||
case simdjson::dom::element_type::OBJECT:
|
||||
for (auto f : simdjson::dom::object(e))
|
||||
{
|
||||
st.str += f.key.size();
|
||||
walk(f.value, st);
|
||||
}
|
||||
break;
|
||||
case simdjson::dom::element_type::ARRAY:
|
||||
for (auto c : simdjson::dom::array(e))
|
||||
{
|
||||
walk(c, st);
|
||||
}
|
||||
break;
|
||||
case simdjson::dom::element_type::STRING:
|
||||
st.str += std::string_view(e).size();
|
||||
break;
|
||||
case simdjson::dom::element_type::INT64:
|
||||
st.num += static_cast<double>(int64_t(e));
|
||||
break;
|
||||
case simdjson::dom::element_type::UINT64:
|
||||
st.num += static_cast<double>(uint64_t(e));
|
||||
break;
|
||||
case simdjson::dom::element_type::DOUBLE:
|
||||
st.num += double(e);
|
||||
break;
|
||||
default:
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
#if JSON_VIEW_BENCH_BOOST
|
||||
static void walk(const boost::json::value& v, stats& st)
|
||||
{
|
||||
++st.nodes;
|
||||
switch (v.kind())
|
||||
{
|
||||
case boost::json::kind::object:
|
||||
for (const auto& kv : v.get_object())
|
||||
{
|
||||
st.str += kv.key().size();
|
||||
walk(kv.value(), st);
|
||||
}
|
||||
break;
|
||||
case boost::json::kind::array:
|
||||
for (const auto& c : v.get_array())
|
||||
{
|
||||
walk(c, st);
|
||||
}
|
||||
break;
|
||||
case boost::json::kind::string:
|
||||
st.str += v.get_string().size();
|
||||
break;
|
||||
case boost::json::kind::int64:
|
||||
st.num += static_cast<double>(v.get_int64());
|
||||
break;
|
||||
case boost::json::kind::uint64:
|
||||
st.num += static_cast<double>(v.get_uint64());
|
||||
break;
|
||||
case boost::json::kind::double_:
|
||||
st.num += v.get_double();
|
||||
break;
|
||||
default:
|
||||
break;
|
||||
}
|
||||
}
|
||||
#endif
|
||||
|
||||
static std::string slurp(const std::string& p)
|
||||
{
|
||||
std::ifstream f(p, std::ios::binary);
|
||||
std::stringstream ss;
|
||||
ss << f.rdbuf();
|
||||
return ss.str();
|
||||
}
|
||||
|
||||
static bool same(const stats& a, const stats& b)
|
||||
{
|
||||
return a.nodes == b.nodes && a.str == b.str && (a.num == b.num || std::fabs(a.num - b.num) <= 1e-9 * std::fabs(a.num));
|
||||
}
|
||||
|
||||
int main(int argc, char** argv)
|
||||
{
|
||||
int rounds = 0; // 0: by size
|
||||
std::vector<std::string> files;
|
||||
for (int i = 1; i < argc; ++i)
|
||||
{
|
||||
if (std::strcmp(argv[i], "--rounds") == 0 && i + 1 < argc)
|
||||
{
|
||||
rounds = std::atoi(argv[++i]);
|
||||
}
|
||||
else
|
||||
{
|
||||
files.push_back(argv[i]);
|
||||
}
|
||||
}
|
||||
std::FILE* csv = std::fopen("bench_corpus.csv", "w");
|
||||
std::fprintf(csv, "file,bytes,workload,engine,ns\n");
|
||||
simdjson::dom::parser sj;
|
||||
for (const auto& path : files)
|
||||
{
|
||||
const std::string s = slurp(path);
|
||||
const std::string name = path.substr(path.rfind('/') + 1);
|
||||
const simdjson::padded_string ps(s);
|
||||
|
||||
// all engines must agree before timing
|
||||
stats a, b, c, d;
|
||||
const json_document doc = json_document::parse(s);
|
||||
walk(doc.root(), a);
|
||||
yyjson_doc* y = yyjson_read(s.data(), s.size(), 0);
|
||||
auto sjr = sj.parse(ps);
|
||||
#if JSON_VIEW_BENCH_BOOST
|
||||
boost::json::parse_options opt;
|
||||
opt.numbers = boost::json::number_precision::precise;
|
||||
boost::json::monotonic_resource mr0;
|
||||
const boost::json::value bv = boost::json::parse(s, &mr0, opt);
|
||||
#endif
|
||||
if (y == nullptr || sjr.error())
|
||||
{
|
||||
std::printf("%-34s skipped (an engine rejects it)\n", name.c_str());
|
||||
yyjson_doc_free(y);
|
||||
continue;
|
||||
}
|
||||
walk(yyjson_doc_get_root(y), b);
|
||||
walk(sjr.value_unsafe(), c);
|
||||
#if JSON_VIEW_BENCH_BOOST
|
||||
walk(bv, d);
|
||||
#else
|
||||
d = a;
|
||||
#endif
|
||||
yyjson_doc_free(y);
|
||||
const bool ok = same(a, b) && same(a, c) && same(a, d);
|
||||
|
||||
const int r = rounds > 0 ? rounds : static_cast<int>(std::max<std::size_t>(3, std::min<std::size_t>(60, 400000000 / (s.size() + 1))));
|
||||
struct engine
|
||||
{
|
||||
std::string name;
|
||||
std::function<void()> fn;
|
||||
};
|
||||
json_document vd = json_document::parse(s);
|
||||
yyjson_doc* yd = yyjson_read(s.data(), s.size(), 0);
|
||||
simdjson::dom::parser sjd;
|
||||
const simdjson::dom::element se = sjd.parse(ps).value_unsafe();
|
||||
const std::vector<std::pair<std::string, std::vector<engine>>> workloads =
|
||||
{
|
||||
{
|
||||
"parse", {
|
||||
{"json_view", [&] { auto x = json_document::parse(s); g_sink = static_cast<double>(x.node_count()); }},
|
||||
{"yyjson", [&] { yyjson_doc* x = yyjson_read(s.data(), s.size(), 0); g_sink = static_cast<double>(yyjson_doc_get_val_count(x)); yyjson_doc_free(x); }},
|
||||
{"simdjson DOM", [&] { auto e = sj.parse(ps).value_unsafe(); g_sink = e.is_object(); }},
|
||||
#if JSON_VIEW_BENCH_BOOST
|
||||
{"Boost.JSON", [&] { boost::json::monotonic_resource mr; auto v = boost::json::parse(s, &mr); g_sink = v.is_object(); }},
|
||||
#endif
|
||||
}
|
||||
},
|
||||
{
|
||||
"traverse", {
|
||||
{"json_view", [&] { auto x = json_document::parse(s); stats st; walk(x.root(), st); g_sink = st.num; }},
|
||||
{"yyjson", [&] { yyjson_doc* x = yyjson_read(s.data(), s.size(), 0); stats st; walk(yyjson_doc_get_root(x), st); g_sink = st.num; yyjson_doc_free(x); }},
|
||||
{"simdjson DOM", [&] { stats st; walk(sj.parse(ps).value_unsafe(), st); g_sink = st.num; }},
|
||||
#if JSON_VIEW_BENCH_BOOST
|
||||
{"Boost.JSON", [&] { boost::json::monotonic_resource mr; auto v = boost::json::parse(s, &mr); stats st; walk(v, st); g_sink = st.num; }},
|
||||
#endif
|
||||
}
|
||||
},
|
||||
{
|
||||
"dump", {
|
||||
{"json_view", [&] { std::string o = vd.root().dump(); g_sink = static_cast<double>(o.size()); }},
|
||||
{"yyjson", [&] { std::size_t n = 0; char* o = yyjson_write(yd, 0, &n); g_sink = static_cast<double>(n); std::free(o); }},
|
||||
{"simdjson DOM", [&] { std::string o = simdjson::to_string(se); g_sink = static_cast<double>(o.size()); }},
|
||||
{"json_view (source numbers)", [&] { std::string o = vd.root().dump(-1, ' ', false, json_view::number_format::source); g_sink = static_cast<double>(o.size()); }},
|
||||
}
|
||||
},
|
||||
};
|
||||
std::printf("%-34s %9zu B%s\n", name.c_str(), s.size(), ok ? "" : " [ENGINES DISAGREE]");
|
||||
for (const auto& wl : workloads)
|
||||
{
|
||||
std::vector<double> best(wl.second.size(), 1e300);
|
||||
for (int i = 0; i < r; ++i)
|
||||
{
|
||||
for (std::size_t k = 0; k < wl.second.size(); ++k)
|
||||
{
|
||||
const auto t0 = std::chrono::steady_clock::now();
|
||||
wl.second[k].fn();
|
||||
best[k] = std::min(best[k], std::chrono::duration<double, std::nano>(std::chrono::steady_clock::now() - t0).count());
|
||||
}
|
||||
}
|
||||
std::printf(" %-9s", wl.first.c_str());
|
||||
for (std::size_t k = 0; k < wl.second.size(); ++k)
|
||||
{
|
||||
std::printf(" %s %.2f GB/s (%.2fx)", wl.second[k].name.c_str(), static_cast<double>(s.size()) / best[k], best[k] / best[0]);
|
||||
std::fprintf(csv, "%s,%zu,%s,%s,%.1f\n", name.c_str(), s.size(), wl.first.c_str(), wl.second[k].name.c_str(), best[k]);
|
||||
}
|
||||
std::printf("\n");
|
||||
std::fflush(stdout);
|
||||
}
|
||||
yyjson_doc_free(yd);
|
||||
}
|
||||
std::fclose(csv);
|
||||
}
|
||||
@@ -0,0 +1,735 @@
|
||||
// __ _____ _____ _____
|
||||
// __| | __| | | | JSON for Modern C++ (supporting code)
|
||||
// | | |__ | | | | | | version 3.12.0
|
||||
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
|
||||
//
|
||||
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
// Same-feature-set benchmark: read-only JSON documents with random access.
|
||||
//
|
||||
// json_view nlohmann/json_view.hpp (fresh document per parse / reused)
|
||||
// yyjson yyjson_read(): immutable document, random access
|
||||
// simdjson DOM dom::parser (reused, as recommended): immutable, random access
|
||||
// references (different feature sets):
|
||||
// simdjson OD On-Demand: forward-only, lazy
|
||||
// Boost.JSON owning, mutable DOM (monotonic resource)
|
||||
// json::parse owning, mutable DOM (nlohmann today)
|
||||
//
|
||||
// Workloads: parse (build + free), traverse (visit everything, convert every
|
||||
// number, touch every string and key), select (a few fields per document),
|
||||
// dump (compact serialization of the parsed document).
|
||||
// All engines run interleaved in every round; the best round is reported.
|
||||
#include <nlohmann/json_view.hpp>
|
||||
|
||||
#if JSON_VIEW_BENCH_BOOST
|
||||
#include <boost/json.hpp>
|
||||
#include <boost/json/src.hpp>
|
||||
#endif
|
||||
#include <simdjson.h>
|
||||
#include <yyjson.h>
|
||||
|
||||
#include <algorithm>
|
||||
#include <chrono>
|
||||
#include <cmath>
|
||||
#include <cstdio>
|
||||
#include <fstream>
|
||||
#include <functional>
|
||||
#include <map>
|
||||
#include <sstream>
|
||||
|
||||
using nlohmann::json;
|
||||
using nlohmann::json_document;
|
||||
using nlohmann::json_view;
|
||||
|
||||
static volatile double g_sink;
|
||||
|
||||
struct stats
|
||||
{
|
||||
double num = 0;
|
||||
std::size_t str = 0, nodes = 0;
|
||||
};
|
||||
|
||||
// ---------------- traversal ----------------
|
||||
|
||||
static void walk(json_view v, stats& st)
|
||||
{
|
||||
++st.nodes;
|
||||
switch (v.type())
|
||||
{
|
||||
case json::value_t::object:
|
||||
for (auto it = v.begin(); it != v.end(); ++it)
|
||||
{
|
||||
st.str += it.key().size();
|
||||
walk(*it, st);
|
||||
}
|
||||
break;
|
||||
case json::value_t::array:
|
||||
for (const json_view e : v)
|
||||
{
|
||||
walk(e, st);
|
||||
}
|
||||
break;
|
||||
case json::value_t::string:
|
||||
st.str += v.get_string().size();
|
||||
break;
|
||||
case json::value_t::number_integer:
|
||||
st.num += static_cast<double>(v.get<std::int64_t>());
|
||||
break;
|
||||
case json::value_t::number_unsigned:
|
||||
st.num += static_cast<double>(v.get<std::uint64_t>());
|
||||
break;
|
||||
case json::value_t::number_float:
|
||||
st.num += v.get<double>();
|
||||
break;
|
||||
default:
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
static void walk(const json& j, stats& st)
|
||||
{
|
||||
++st.nodes;
|
||||
switch (j.type())
|
||||
{
|
||||
case json::value_t::object:
|
||||
for (const auto& kv : j.get_ref<const json::object_t&>())
|
||||
{
|
||||
st.str += kv.first.size();
|
||||
walk(kv.second, st);
|
||||
}
|
||||
break;
|
||||
case json::value_t::array:
|
||||
for (const auto& e : j.get_ref<const json::array_t&>())
|
||||
{
|
||||
walk(e, st);
|
||||
}
|
||||
break;
|
||||
case json::value_t::string:
|
||||
st.str += j.get_ref<const std::string&>().size();
|
||||
break;
|
||||
case json::value_t::number_integer:
|
||||
st.num += static_cast<double>(*j.get_ptr<const json::number_integer_t*>());
|
||||
break;
|
||||
case json::value_t::number_unsigned:
|
||||
st.num += static_cast<double>(*j.get_ptr<const json::number_unsigned_t*>());
|
||||
break;
|
||||
case json::value_t::number_float:
|
||||
st.num += *j.get_ptr<const json::number_float_t*>();
|
||||
break;
|
||||
default:
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
static void walk(yyjson_val* v, stats& st)
|
||||
{
|
||||
++st.nodes;
|
||||
switch (yyjson_get_type(v))
|
||||
{
|
||||
case YYJSON_TYPE_OBJ:
|
||||
{
|
||||
std::size_t idx, max;
|
||||
yyjson_val* k, * val;
|
||||
yyjson_obj_foreach(v, idx, max, k, val)
|
||||
{
|
||||
st.str += yyjson_get_len(k);
|
||||
walk(val, st);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case YYJSON_TYPE_ARR:
|
||||
{
|
||||
std::size_t idx, max;
|
||||
yyjson_val* val;
|
||||
yyjson_arr_foreach(v, idx, max, val)
|
||||
{
|
||||
walk(val, st);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case YYJSON_TYPE_STR:
|
||||
st.str += yyjson_get_len(v);
|
||||
break;
|
||||
case YYJSON_TYPE_NUM:
|
||||
if (yyjson_is_sint(v))
|
||||
{
|
||||
st.num += static_cast<double>(yyjson_get_sint(v));
|
||||
}
|
||||
else if (yyjson_is_uint(v))
|
||||
{
|
||||
st.num += static_cast<double>(yyjson_get_uint(v));
|
||||
}
|
||||
else
|
||||
{
|
||||
st.num += yyjson_get_real(v);
|
||||
}
|
||||
break;
|
||||
default:
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
static void walk(simdjson::dom::element e, stats& st)
|
||||
{
|
||||
++st.nodes;
|
||||
switch (e.type())
|
||||
{
|
||||
case simdjson::dom::element_type::OBJECT:
|
||||
for (auto f : simdjson::dom::object(e))
|
||||
{
|
||||
st.str += f.key.size();
|
||||
walk(f.value, st);
|
||||
}
|
||||
break;
|
||||
case simdjson::dom::element_type::ARRAY:
|
||||
for (auto c : simdjson::dom::array(e))
|
||||
{
|
||||
walk(c, st);
|
||||
}
|
||||
break;
|
||||
case simdjson::dom::element_type::STRING:
|
||||
st.str += std::string_view(e).size();
|
||||
break;
|
||||
case simdjson::dom::element_type::INT64:
|
||||
st.num += static_cast<double>(int64_t(e));
|
||||
break;
|
||||
case simdjson::dom::element_type::UINT64:
|
||||
st.num += static_cast<double>(uint64_t(e));
|
||||
break;
|
||||
case simdjson::dom::element_type::DOUBLE:
|
||||
st.num += double(e);
|
||||
break;
|
||||
default:
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
static void walk_od(simdjson::ondemand::value v, stats& st)
|
||||
{
|
||||
++st.nodes;
|
||||
switch (v.type())
|
||||
{
|
||||
case simdjson::ondemand::json_type::object:
|
||||
for (auto f : v.get_object())
|
||||
{
|
||||
st.str += std::string_view(f.unescaped_key()).size();
|
||||
walk_od(f.value(), st);
|
||||
}
|
||||
break;
|
||||
case simdjson::ondemand::json_type::array:
|
||||
for (auto c : v.get_array())
|
||||
{
|
||||
walk_od(c.value(), st);
|
||||
}
|
||||
break;
|
||||
case simdjson::ondemand::json_type::string:
|
||||
st.str += std::string_view(v.get_string()).size();
|
||||
break;
|
||||
case simdjson::ondemand::json_type::number:
|
||||
{
|
||||
simdjson::ondemand::number n = v.get_number();
|
||||
switch (n.get_number_type())
|
||||
{
|
||||
case simdjson::ondemand::number_type::signed_integer:
|
||||
st.num += static_cast<double>(n.get_int64());
|
||||
break;
|
||||
case simdjson::ondemand::number_type::unsigned_integer:
|
||||
st.num += static_cast<double>(n.get_uint64());
|
||||
break;
|
||||
default:
|
||||
st.num += n.get_double();
|
||||
break;
|
||||
}
|
||||
break;
|
||||
}
|
||||
case simdjson::ondemand::json_type::boolean:
|
||||
(void)bool(v.get_bool());
|
||||
break;
|
||||
default:
|
||||
(void)v.is_null();
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
#if JSON_VIEW_BENCH_BOOST
|
||||
static void walk(const boost::json::value& v, stats& st)
|
||||
{
|
||||
++st.nodes;
|
||||
switch (v.kind())
|
||||
{
|
||||
case boost::json::kind::object:
|
||||
for (const auto& kv : v.get_object())
|
||||
{
|
||||
st.str += kv.key().size();
|
||||
walk(kv.value(), st);
|
||||
}
|
||||
break;
|
||||
case boost::json::kind::array:
|
||||
for (const auto& c : v.get_array())
|
||||
{
|
||||
walk(c, st);
|
||||
}
|
||||
break;
|
||||
case boost::json::kind::string:
|
||||
st.str += v.get_string().size();
|
||||
break;
|
||||
case boost::json::kind::int64:
|
||||
st.num += static_cast<double>(v.get_int64());
|
||||
break;
|
||||
case boost::json::kind::uint64:
|
||||
st.num += static_cast<double>(v.get_uint64());
|
||||
break;
|
||||
case boost::json::kind::double_:
|
||||
st.num += v.get_double();
|
||||
break;
|
||||
default:
|
||||
break;
|
||||
}
|
||||
}
|
||||
#endif
|
||||
|
||||
// ---------------- selective access ----------------
|
||||
// twitter: per status id, user.screen_name, retweet_count
|
||||
// citm: per performance id, eventId, #seatCategories; #events
|
||||
// canada: type, features[0].geometry.type, #coordinates
|
||||
// jeopardy: per question: round == "Final Jeopardy!", len(category)
|
||||
// status (one tweet): id, user.screen_name, retweet_count
|
||||
// rpc: method, params.minuend, id
|
||||
|
||||
static double pick(const std::string& name, json_view r)
|
||||
{
|
||||
double acc = 0;
|
||||
if (name == "twitter" || name == "status")
|
||||
{
|
||||
auto one = [&](json_view s)
|
||||
{
|
||||
acc += static_cast<double>(s["id"].get<std::uint64_t>());
|
||||
acc += static_cast<double>(s["user"]["screen_name"].get_string().size());
|
||||
acc += static_cast<double>(s["retweet_count"].get<std::int64_t>());
|
||||
};
|
||||
if (name == "twitter")
|
||||
{
|
||||
for (const json_view s : r["statuses"])
|
||||
{
|
||||
one(s);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
one(r);
|
||||
}
|
||||
}
|
||||
else if (name == "citm_catalog")
|
||||
{
|
||||
for (const json_view p : r["performances"])
|
||||
{
|
||||
acc += static_cast<double>(p["id"].get<std::uint64_t>() + p["eventId"].get<std::uint64_t>() + p["seatCategories"].size());
|
||||
}
|
||||
acc += static_cast<double>(r["events"].size());
|
||||
}
|
||||
else if (name == "canada")
|
||||
{
|
||||
const json_view g = r["features"][0]["geometry"];
|
||||
acc += static_cast<double>(r["type"].get_string().size() + g["type"].get_string().size() + g["coordinates"].size());
|
||||
}
|
||||
else if (name == "jeopardy")
|
||||
{
|
||||
for (const json_view q : r)
|
||||
{
|
||||
acc += q["round"].get_string() == "Final Jeopardy!" ? 1 : 0;
|
||||
acc += static_cast<double>(q["category"].get_string().size());
|
||||
}
|
||||
}
|
||||
else if (name == "rpc")
|
||||
{
|
||||
acc += static_cast<double>(r["method"].get_string().size());
|
||||
acc += static_cast<double>(r["params"]["minuend"].get<std::int64_t>() + r["id"].get<std::int64_t>());
|
||||
}
|
||||
return acc;
|
||||
}
|
||||
|
||||
static double pick(const std::string& name, const json& r)
|
||||
{
|
||||
double acc = 0;
|
||||
if (name == "twitter" || name == "status")
|
||||
{
|
||||
auto one = [&](const json & s)
|
||||
{
|
||||
acc += static_cast<double>(s["id"].get<std::uint64_t>());
|
||||
acc += static_cast<double>(s["user"]["screen_name"].get_ref<const std::string&>().size());
|
||||
acc += static_cast<double>(s["retweet_count"].get<std::int64_t>());
|
||||
};
|
||||
if (name == "twitter")
|
||||
{
|
||||
for (const auto& s : r["statuses"])
|
||||
{
|
||||
one(s);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
one(r);
|
||||
}
|
||||
}
|
||||
else if (name == "citm_catalog")
|
||||
{
|
||||
for (const auto& p : r["performances"])
|
||||
{
|
||||
acc += static_cast<double>(p["id"].get<std::uint64_t>() + p["eventId"].get<std::uint64_t>() + p["seatCategories"].size());
|
||||
}
|
||||
acc += static_cast<double>(r["events"].size());
|
||||
}
|
||||
else if (name == "canada")
|
||||
{
|
||||
const json& g = r["features"][0]["geometry"];
|
||||
acc += static_cast<double>(r["type"].get_ref<const std::string&>().size() + g["type"].get_ref<const std::string&>().size() + g["coordinates"].size());
|
||||
}
|
||||
else if (name == "jeopardy")
|
||||
{
|
||||
for (const auto& q : r)
|
||||
{
|
||||
acc += q["round"].get_ref<const std::string&>() == "Final Jeopardy!" ? 1 : 0;
|
||||
acc += static_cast<double>(q["category"].get_ref<const std::string&>().size());
|
||||
}
|
||||
}
|
||||
else if (name == "rpc")
|
||||
{
|
||||
acc += static_cast<double>(r["method"].get_ref<const std::string&>().size());
|
||||
acc += static_cast<double>(r["params"]["minuend"].get<std::int64_t>() + r["id"].get<std::int64_t>());
|
||||
}
|
||||
return acc;
|
||||
}
|
||||
|
||||
static double pick(const std::string& name, yyjson_val* r)
|
||||
{
|
||||
double acc = 0;
|
||||
auto get = [](yyjson_val * o, const char* k)
|
||||
{
|
||||
return yyjson_obj_get(o, k);
|
||||
};
|
||||
if (name == "twitter" || name == "status")
|
||||
{
|
||||
auto one = [&](yyjson_val * s)
|
||||
{
|
||||
acc += static_cast<double>(yyjson_get_uint(get(s, "id")));
|
||||
acc += static_cast<double>(yyjson_get_len(get(get(s, "user"), "screen_name")));
|
||||
acc += static_cast<double>(yyjson_get_sint(get(s, "retweet_count")));
|
||||
};
|
||||
if (name == "twitter")
|
||||
{
|
||||
std::size_t idx, max;
|
||||
yyjson_val* s;
|
||||
yyjson_arr_foreach(get(r, "statuses"), idx, max, s)
|
||||
{
|
||||
one(s);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
one(r);
|
||||
}
|
||||
}
|
||||
else if (name == "citm_catalog")
|
||||
{
|
||||
std::size_t idx, max;
|
||||
yyjson_val* p;
|
||||
yyjson_arr_foreach(get(r, "performances"), idx, max, p)
|
||||
{
|
||||
acc += static_cast<double>(yyjson_get_uint(get(p, "id")) + yyjson_get_uint(get(p, "eventId")) + yyjson_arr_size(get(p, "seatCategories")));
|
||||
}
|
||||
acc += static_cast<double>(yyjson_obj_size(get(r, "events")));
|
||||
}
|
||||
else if (name == "canada")
|
||||
{
|
||||
yyjson_val* g = get(yyjson_arr_get(get(r, "features"), 0), "geometry");
|
||||
acc += static_cast<double>(yyjson_get_len(get(r, "type")) + yyjson_get_len(get(g, "type")) + yyjson_arr_size(get(g, "coordinates")));
|
||||
}
|
||||
else if (name == "jeopardy")
|
||||
{
|
||||
std::size_t idx, max;
|
||||
yyjson_val* q;
|
||||
yyjson_arr_foreach(r, idx, max, q)
|
||||
{
|
||||
acc += yyjson_equals_str(get(q, "round"), "Final Jeopardy!") ? 1 : 0;
|
||||
acc += static_cast<double>(yyjson_get_len(get(q, "category")));
|
||||
}
|
||||
}
|
||||
else if (name == "rpc")
|
||||
{
|
||||
acc += static_cast<double>(yyjson_get_len(get(r, "method")));
|
||||
acc += static_cast<double>(yyjson_get_sint(get(get(r, "params"), "minuend")) + yyjson_get_sint(get(r, "id")));
|
||||
}
|
||||
return acc;
|
||||
}
|
||||
|
||||
static double pick(const std::string& name, simdjson::dom::element r)
|
||||
{
|
||||
double acc = 0;
|
||||
if (name == "twitter" || name == "status")
|
||||
{
|
||||
auto one = [&](simdjson::dom::element s)
|
||||
{
|
||||
acc += static_cast<double>(uint64_t(s["id"]));
|
||||
acc += static_cast<double>(std::string_view(s["user"]["screen_name"]).size());
|
||||
acc += static_cast<double>(int64_t(s["retweet_count"]));
|
||||
};
|
||||
if (name == "twitter")
|
||||
{
|
||||
for (auto s : simdjson::dom::array(r["statuses"]))
|
||||
{
|
||||
one(s);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
one(r);
|
||||
}
|
||||
}
|
||||
else if (name == "citm_catalog")
|
||||
{
|
||||
for (auto p : simdjson::dom::array(r["performances"]))
|
||||
{
|
||||
acc += static_cast<double>(uint64_t(p["id"]) + uint64_t(p["eventId"]) + simdjson::dom::array(p["seatCategories"]).size());
|
||||
}
|
||||
acc += static_cast<double>(simdjson::dom::object(r["events"]).size());
|
||||
}
|
||||
else if (name == "canada")
|
||||
{
|
||||
auto g = r["features"].at(0)["geometry"];
|
||||
acc += static_cast<double>(std::string_view(r["type"]).size() + std::string_view(g["type"]).size() + simdjson::dom::array(g["coordinates"]).size());
|
||||
}
|
||||
else if (name == "jeopardy")
|
||||
{
|
||||
for (auto q : simdjson::dom::array(r))
|
||||
{
|
||||
acc += std::string_view(q["round"]) == "Final Jeopardy!" ? 1 : 0;
|
||||
acc += static_cast<double>(std::string_view(q["category"]).size());
|
||||
}
|
||||
}
|
||||
else if (name == "rpc")
|
||||
{
|
||||
acc += static_cast<double>(std::string_view(r["method"]).size());
|
||||
acc += static_cast<double>(int64_t(r["params"]["minuend"]) + int64_t(r["id"]));
|
||||
}
|
||||
return acc;
|
||||
}
|
||||
|
||||
static double pick_od(const std::string& name, simdjson::ondemand::document& d)
|
||||
{
|
||||
double acc = 0;
|
||||
if (name == "twitter" || name == "status")
|
||||
{
|
||||
auto one = [&](simdjson::ondemand::object s)
|
||||
{
|
||||
acc += static_cast<double>(uint64_t(s["id"]));
|
||||
acc += static_cast<double>(std::string_view(s["user"]["screen_name"]).size());
|
||||
acc += static_cast<double>(int64_t(s["retweet_count"]));
|
||||
};
|
||||
if (name == "twitter")
|
||||
{
|
||||
for (auto s : d["statuses"])
|
||||
{
|
||||
one(s.get_object());
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
one(d.get_object());
|
||||
}
|
||||
}
|
||||
else if (name == "citm_catalog")
|
||||
{
|
||||
simdjson::ondemand::object ev = d["events"].get_object();
|
||||
acc += static_cast<double>(ev.count_fields());
|
||||
for (auto p : d["performances"])
|
||||
{
|
||||
simdjson::ondemand::object o = p.get_object();
|
||||
const auto a = uint64_t(o["eventId"]) + uint64_t(o["id"]);
|
||||
simdjson::ondemand::array sc = o["seatCategories"].get_array();
|
||||
acc += static_cast<double>(a + sc.count_elements());
|
||||
}
|
||||
}
|
||||
else if (name == "canada")
|
||||
{
|
||||
acc += static_cast<double>(std::string_view(d["type"]).size());
|
||||
auto g = d["features"].at(0)["geometry"];
|
||||
acc += static_cast<double>(std::string_view(g["type"]).size());
|
||||
simdjson::ondemand::array co = g["coordinates"].get_array();
|
||||
acc += static_cast<double>(co.count_elements());
|
||||
}
|
||||
else if (name == "jeopardy")
|
||||
{
|
||||
for (auto q : d)
|
||||
{
|
||||
simdjson::ondemand::object o = q.get_object();
|
||||
acc += static_cast<double>(std::string_view(o["category"]).size());
|
||||
acc += std::string_view(o["round"]) == "Final Jeopardy!" ? 1 : 0;
|
||||
}
|
||||
}
|
||||
else if (name == "rpc")
|
||||
{
|
||||
acc += static_cast<double>(std::string_view(d["method"]).size());
|
||||
acc += static_cast<double>(int64_t(d["params"]["minuend"]));
|
||||
acc += static_cast<double>(int64_t(d["id"]));
|
||||
}
|
||||
return acc;
|
||||
}
|
||||
|
||||
// ---------------- harness ----------------
|
||||
|
||||
static std::string slurp(const std::string& p)
|
||||
{
|
||||
std::ifstream f(p, std::ios::binary);
|
||||
std::stringstream ss;
|
||||
ss << f.rdbuf();
|
||||
return ss.str();
|
||||
}
|
||||
|
||||
struct engine
|
||||
{
|
||||
std::string name;
|
||||
std::function<void()> fn;
|
||||
};
|
||||
|
||||
int main(int argc, char** argv)
|
||||
{
|
||||
if (argc < 2)
|
||||
{
|
||||
std::fprintf(stderr, "usage: %s <json_test_data directory> [rounds] [document]\n", argv[0]);
|
||||
return 1;
|
||||
}
|
||||
const std::string T = std::string(argv[1]) + "/";
|
||||
const int rounds = argc > 2 ? std::atoi(argv[2]) : 30;
|
||||
const std::string only = argc > 3 ? argv[3] : "";
|
||||
struct doc
|
||||
{
|
||||
std::string name, text;
|
||||
int batch;
|
||||
};
|
||||
std::vector<doc> docs;
|
||||
for (const char* f :
|
||||
{"nativejson-benchmark/twitter.json", "nativejson-benchmark/citm_catalog.json", "nativejson-benchmark/canada.json", "jeopardy/jeopardy.json"
|
||||
})
|
||||
{
|
||||
std::string n = std::string(f).substr(std::string(f).find('/') + 1);
|
||||
docs.push_back({n.substr(0, n.size() - 5), slurp(T + f), 1});
|
||||
}
|
||||
docs.push_back({"status", json::parse(docs[0].text)["statuses"][0].dump(), 200});
|
||||
docs.push_back({"rpc", R"({"jsonrpc": "2.0", "method": "subtract", "params": {"minuend": 42, "subtrahend": 23}, "id": 3})", 5000});
|
||||
|
||||
// correctness cross-check of the workloads
|
||||
for (const auto& dc : docs)
|
||||
{
|
||||
stats a, b, c, dd;
|
||||
walk(json::parse(dc.text), a);
|
||||
auto d = json_document::parse(dc.text);
|
||||
walk(d.root(), b);
|
||||
yyjson_doc* y = yyjson_read(dc.text.data(), dc.text.size(), 0);
|
||||
walk(yyjson_doc_get_root(y), c);
|
||||
simdjson::dom::parser p;
|
||||
walk(p.parse(dc.text).value(), dd);
|
||||
const bool ok = a.nodes == b.nodes && a.nodes == c.nodes && a.nodes == dd.nodes && a.str == b.str && a.str == c.str && a.str == dd.str
|
||||
&& std::fabs(a.num - b.num) <= 1e-9 * std::fabs(a.num) && std::fabs(a.num - c.num) <= 1e-9 * std::fabs(a.num);
|
||||
const double pa = pick(dc.name, json::parse(dc.text)), pb = pick(dc.name, d.root()), pc = pick(dc.name, yyjson_doc_get_root(y)), pd = pick(dc.name, p.parse(dc.text).value());
|
||||
std::printf("check %-13s traverse %s select %s\n", dc.name.c_str(), ok ? "OK" : "MISMATCH", (pa == pb && pa == pc && pa == pd) ? "OK" : "MISMATCH");
|
||||
yyjson_doc_free(y);
|
||||
}
|
||||
|
||||
std::FILE* csv = std::fopen("bench_view.csv", "w");
|
||||
std::fprintf(csv, "doc,bytes,workload,engine,ns\n");
|
||||
json_document reused;
|
||||
simdjson::dom::parser sj;
|
||||
simdjson::ondemand::parser od;
|
||||
for (const auto& dc : docs)
|
||||
{
|
||||
if (!only.empty() && dc.name != only)
|
||||
{
|
||||
continue;
|
||||
}
|
||||
const std::string& s = dc.text;
|
||||
const simdjson::padded_string ps(s);
|
||||
const std::string name = dc.name;
|
||||
std::vector<std::pair<std::string, std::vector<engine>>> workloads;
|
||||
|
||||
workloads.push_back({"parse", {
|
||||
{"json_view", [&] { auto d = json_document::parse(s); g_sink = static_cast<double>(d.node_count()); }},
|
||||
{"json_view (reused)", [&] { reused.read(s); g_sink = static_cast<double>(reused.node_count()); }},
|
||||
{"yyjson", [&] { yyjson_doc* d = yyjson_read(s.data(), s.size(), 0); g_sink = static_cast<double>(yyjson_doc_get_val_count(d)); yyjson_doc_free(d); }},
|
||||
{"simdjson DOM", [&] { auto e = sj.parse(ps).value_unsafe(); g_sink = e.is_object(); }},
|
||||
#if JSON_VIEW_BENCH_BOOST
|
||||
{"Boost.JSON", [&] { boost::json::monotonic_resource mr; auto v = boost::json::parse(s, &mr); g_sink = v.is_object(); }},
|
||||
#endif
|
||||
{"json::parse", [&] { json j = json::parse(s); g_sink = static_cast<double>(j.size()); }},
|
||||
}});
|
||||
workloads.push_back({"traverse", {
|
||||
{"json_view", [&] { auto d = json_document::parse(s); stats st; walk(d.root(), st); g_sink = st.num; }},
|
||||
{"yyjson", [&] { yyjson_doc* d = yyjson_read(s.data(), s.size(), 0); stats st; walk(yyjson_doc_get_root(d), st); g_sink = st.num; yyjson_doc_free(d); }},
|
||||
{"simdjson DOM", [&] { stats st; walk(sj.parse(ps).value_unsafe(), st); g_sink = st.num; }},
|
||||
{"simdjson OD", [&] { auto d = od.iterate(ps).value_unsafe(); stats st; walk_od(d.get_value().value_unsafe(), st); g_sink = st.num; }},
|
||||
#if JSON_VIEW_BENCH_BOOST
|
||||
{"Boost.JSON", [&] { boost::json::monotonic_resource mr; auto v = boost::json::parse(s, &mr); stats st; walk(v, st); g_sink = st.num; }},
|
||||
#endif
|
||||
{"json::parse", [&] { json j = json::parse(s); stats st; walk(j, st); g_sink = st.num; }},
|
||||
}});
|
||||
workloads.push_back({"select", {
|
||||
{"json_view", [&] { auto d = json_document::parse(s); g_sink = pick(name, d.root()); }},
|
||||
{"yyjson", [&] { yyjson_doc* d = yyjson_read(s.data(), s.size(), 0); g_sink = pick(name, yyjson_doc_get_root(d)); yyjson_doc_free(d); }},
|
||||
{"simdjson DOM", [&] { g_sink = pick(name, sj.parse(ps).value_unsafe()); }},
|
||||
{"simdjson OD", [&] { auto d = od.iterate(ps).value_unsafe(); g_sink = pick_od(name, d); }},
|
||||
{"json::parse", [&] { json j = json::parse(s); g_sink = pick(name, j); }},
|
||||
}});
|
||||
{
|
||||
// serialization of an already parsed document
|
||||
static json_document vd;
|
||||
vd.read(s);
|
||||
static yyjson_doc* yd = nullptr;
|
||||
if (yd)
|
||||
{
|
||||
yyjson_doc_free(yd);
|
||||
}
|
||||
yd = yyjson_read(s.data(), s.size(), 0);
|
||||
static simdjson::dom::parser sjd;
|
||||
static simdjson::dom::element se;
|
||||
se = sjd.parse(ps).value_unsafe();
|
||||
static json jd;
|
||||
jd = json::parse(s);
|
||||
workloads.push_back({"dump", {
|
||||
{"json_view", [&] { std::string o = vd.root().dump(); g_sink = static_cast<double>(o.size()); }},
|
||||
{"yyjson", [&] { std::size_t n = 0; char* o = yyjson_write(yd, 0, &n); g_sink = static_cast<double>(n); std::free(o); }},
|
||||
{"simdjson DOM", [&] { std::string o = simdjson::to_string(se); g_sink = static_cast<double>(o.size()); }},
|
||||
{"json::parse", [&] { std::string o = jd.dump(); g_sink = static_cast<double>(o.size()); }},
|
||||
}});
|
||||
}
|
||||
|
||||
for (auto& wl : workloads)
|
||||
{
|
||||
std::vector<double> best(wl.second.size(), 1e300);
|
||||
const int r = s.size() > 10000000 ? std::max(3, rounds / 5) : rounds;
|
||||
for (int i = 0; i < r; ++i)
|
||||
{
|
||||
for (std::size_t k = 0; k < wl.second.size(); ++k)
|
||||
{
|
||||
const auto t0 = std::chrono::steady_clock::now();
|
||||
for (int b = 0; b < dc.batch; ++b)
|
||||
{
|
||||
wl.second[k].fn();
|
||||
}
|
||||
const double ns = std::chrono::duration<double, std::nano>(std::chrono::steady_clock::now() - t0).count() / dc.batch;
|
||||
best[k] = std::min(best[k], ns);
|
||||
}
|
||||
}
|
||||
const double ref = best[0];
|
||||
std::printf("%-13s %-9s", dc.name.c_str(), wl.first.c_str());
|
||||
for (std::size_t k = 0; k < wl.second.size(); ++k)
|
||||
{
|
||||
const double us = best[k] / 1e3;
|
||||
std::printf(" %s %s%s (%.2fx)", wl.second[k].name.c_str(), us >= 100 ? "" : "", (us >= 1000 ? std::to_string(static_cast<long>(us)) + "us" : (std::to_string(us).substr(0, 5) + "us")).c_str(), best[k] / ref);
|
||||
std::fprintf(csv, "%s,%zu,%s,%s,%.1f\n", dc.name.c_str(), s.size(), wl.first.c_str(), wl.second[k].name.c_str(), best[k]);
|
||||
}
|
||||
std::printf("\n");
|
||||
std::fflush(stdout);
|
||||
}
|
||||
}
|
||||
std::fclose(csv);
|
||||
}
|
||||
Executable
+300
@@ -0,0 +1,300 @@
|
||||
#!/usr/bin/env python3
|
||||
# __ _____ _____ _____
|
||||
# __| | __| | | | JSON for Modern C++ (supporting code)
|
||||
# | | |__ | | | | | | version 3.12.0
|
||||
# |_____|_____|_____|_|___| https://github.com/nlohmann/json
|
||||
#
|
||||
# SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
|
||||
# SPDX-License-Identifier: MIT
|
||||
|
||||
"""Compare json_view with yyjson, simdjson, Boost.JSON, and json::parse.
|
||||
|
||||
Builds bench_view.cpp and bench_corpus.cpp against the include/ directory of
|
||||
this checkout, runs them, and writes the results with everything needed to
|
||||
reproduce them (date, commit, CPU, OS, compiler, library versions, flags) to
|
||||
results/<date>-<host>.md and .csv next to this script.
|
||||
|
||||
The other libraries come from the system (--system, the default: pkg-config
|
||||
or Homebrew) or are downloaded as pinned releases and checked against their
|
||||
SHA-256 (--download). Boost.JSON is optional: without Boost headers, its
|
||||
columns are skipped, and the results say so.
|
||||
|
||||
Only the Python 3 standard library is used; a C++17 compiler is needed.
|
||||
"""
|
||||
|
||||
import argparse
|
||||
import datetime
|
||||
import hashlib
|
||||
import os
|
||||
import platform
|
||||
import re
|
||||
import shlex
|
||||
import shutil
|
||||
import subprocess
|
||||
import sys
|
||||
import tarfile
|
||||
import urllib.request
|
||||
|
||||
HERE = os.path.dirname(os.path.abspath(__file__))
|
||||
REPO = os.path.abspath(os.path.join(HERE, '..', '..', '..'))
|
||||
|
||||
# pinned releases for --download; the hashes are those of the archives
|
||||
PINNED = {
|
||||
'yyjson': {
|
||||
'version': '0.13.0',
|
||||
'url': 'https://github.com/ibireme/yyjson/archive/refs/tags/0.13.0.tar.gz',
|
||||
'sha256': '34e0f62a2bc11ab20d601e8ca1cc2b2079503aa45119a19133d89d19b94a0fae',
|
||||
'dir': 'yyjson-0.13.0',
|
||||
},
|
||||
'simdjson': {
|
||||
'version': '4.6.11',
|
||||
'url': 'https://github.com/simdjson/simdjson/archive/refs/tags/v4.6.11.tar.gz',
|
||||
'sha256': '61d948fc24f0d793829ad658058e7597d064988a89b4607ea02e401a82df98ff',
|
||||
'dir': 'simdjson-4.6.11',
|
||||
},
|
||||
'boost': {
|
||||
'version': '1.92.0',
|
||||
'url': 'https://archives.boost.io/release/1.92.0/source/boost_1_92_0.tar.gz',
|
||||
'sha256': 'c4a3b310ddd2472416e091067166b0713be97c63f38c212c484ada022fd296ce',
|
||||
'dir': 'boost_1_92_0',
|
||||
},
|
||||
}
|
||||
|
||||
# the documents of bench_view.cpp, relative to the json_test_data directory
|
||||
DEFAULT_CORPUS = [
|
||||
'nativejson-benchmark/twitter.json',
|
||||
'nativejson-benchmark/citm_catalog.json',
|
||||
'nativejson-benchmark/canada.json',
|
||||
'jeopardy/jeopardy.json',
|
||||
]
|
||||
|
||||
|
||||
def run(cmd, **kwargs):
|
||||
print('+ ' + ' '.join(shlex.quote(c) for c in cmd), flush=True)
|
||||
return subprocess.run(cmd, check=True, **kwargs)
|
||||
|
||||
|
||||
def output(cmd):
|
||||
try:
|
||||
return subprocess.run(cmd, check=True, capture_output=True, text=True).stdout.strip()
|
||||
except (OSError, subprocess.CalledProcessError):
|
||||
return ''
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# libraries
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class Library:
|
||||
"""include directories, sources to compile, and linker flags of a library"""
|
||||
|
||||
def __init__(self, name, include=None, sources=None, link=None, version=''):
|
||||
self.name = name
|
||||
self.include = include or []
|
||||
self.sources = sources or []
|
||||
self.link = link or []
|
||||
self.version = version
|
||||
|
||||
|
||||
def header_version(path, pattern):
|
||||
try:
|
||||
with open(path, encoding='utf-8', errors='replace') as f:
|
||||
m = re.search(pattern, f.read())
|
||||
return m.group(1) if m else ''
|
||||
except OSError:
|
||||
return ''
|
||||
|
||||
|
||||
def library_version(name, include_dirs):
|
||||
patterns = {
|
||||
'yyjson': ('yyjson.h', r'#define\s+YYJSON_VERSION_STRING\s+"([^"]+)"'),
|
||||
'simdjson': ('simdjson.h', r'#define\s+SIMDJSON_VERSION\s+"?([0-9.]+)"?'),
|
||||
'boost': (os.path.join('boost', 'version.hpp'), r'#define\s+BOOST_LIB_VERSION\s+"([^"]+)"'),
|
||||
}
|
||||
header, pattern = patterns[name]
|
||||
for d in include_dirs:
|
||||
v = header_version(os.path.join(d, header), pattern)
|
||||
if v:
|
||||
return v.replace('_', '.')
|
||||
return ''
|
||||
|
||||
|
||||
def system_library(name):
|
||||
"""a library found with pkg-config or Homebrew, or None"""
|
||||
flags = output(['pkg-config', '--cflags', '--libs', name]).split()
|
||||
if flags:
|
||||
include = [f[2:] for f in flags if f.startswith('-I')]
|
||||
link = [f for f in flags if f.startswith('-L') or f.startswith('-l')]
|
||||
libdirs = [f[2:] for f in link if f.startswith('-L')]
|
||||
link += ['-Wl,-rpath,' + d for d in libdirs]
|
||||
return Library(name, include, [], link, library_version(name, include))
|
||||
prefix = output(['brew', '--prefix', name]) if shutil.which('brew') else ''
|
||||
if prefix and os.path.isdir(os.path.join(prefix, 'include')):
|
||||
include = [os.path.join(prefix, 'include')]
|
||||
link = []
|
||||
if name != 'boost':
|
||||
lib = os.path.join(prefix, 'lib')
|
||||
link = ['-L' + lib, '-l' + name, '-Wl,-rpath,' + lib]
|
||||
return Library(name, include, [], link, library_version(name, include))
|
||||
if name == 'boost':
|
||||
for d in ['/usr/include', '/usr/local/include']:
|
||||
if os.path.isfile(os.path.join(d, 'boost', 'json.hpp')):
|
||||
return Library(name, [d], [], [], library_version(name, [d]))
|
||||
return None
|
||||
|
||||
|
||||
def download_library(name, work):
|
||||
"""a pinned release, downloaded and checked, or an error"""
|
||||
pin = PINNED[name]
|
||||
archive = os.path.join(work, 'download', os.path.basename(pin['url']))
|
||||
os.makedirs(os.path.dirname(archive), exist_ok=True)
|
||||
if not os.path.isfile(archive):
|
||||
print(f'downloading {pin["url"]}', flush=True)
|
||||
urllib.request.urlretrieve(pin['url'], archive)
|
||||
with open(archive, 'rb') as f:
|
||||
digest = hashlib.sha256(f.read()).hexdigest()
|
||||
if digest != pin['sha256']:
|
||||
sys.exit(f'error: SHA-256 of {archive} is {digest}, expected {pin["sha256"]}')
|
||||
src = os.path.join(work, 'download', pin['dir'])
|
||||
if not os.path.isdir(src):
|
||||
with tarfile.open(archive) as t:
|
||||
# (the 'data' filter rejects links and paths outside the target where Python has it)
|
||||
kwargs = {'filter': 'data'} if hasattr(tarfile, 'data_filter') else {}
|
||||
t.extractall(os.path.join(work, 'download'), **kwargs) # noqa: S202 (checked archive)
|
||||
if name == 'yyjson':
|
||||
return Library(name, [os.path.join(src, 'src')], [os.path.join(src, 'src', 'yyjson.c')], [], pin['version'])
|
||||
if name == 'simdjson':
|
||||
single = os.path.join(src, 'singleheader')
|
||||
return Library(name, [single], [os.path.join(single, 'simdjson.cpp')], [], pin['version'])
|
||||
return Library(name, [src], [], [], pin['version'])
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# machine description
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def cpu_model():
|
||||
if sys.platform == 'darwin':
|
||||
return output(['sysctl', '-n', 'machdep.cpu.brand_string'])
|
||||
try:
|
||||
with open('/proc/cpuinfo', encoding='utf-8') as f:
|
||||
for line in f:
|
||||
if line.startswith('model name') or line.startswith('Model'):
|
||||
return line.split(':', 1)[1].strip()
|
||||
except OSError:
|
||||
pass
|
||||
# (AArch64 Linux: /proc/cpuinfo has no model name, lscpu knows it)
|
||||
for line in output(['lscpu']).splitlines():
|
||||
if line.startswith('Model name:'):
|
||||
return line.split(':', 1)[1].strip()
|
||||
return platform.processor() or platform.machine()
|
||||
|
||||
|
||||
def git_commit():
|
||||
commit = output(['git', '-C', REPO, 'rev-parse', '--short=12', 'HEAD'])
|
||||
dirty = output(['git', '-C', REPO, 'status', '--porcelain', '--untracked-files=no'])
|
||||
return commit + (' (with local changes)' if dirty else '')
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# main
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def main():
|
||||
ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
|
||||
ap.add_argument('--data', required=True, help='json_test_data directory (with nativejson-benchmark/ and jeopardy/)')
|
||||
ap.add_argument('--download', action='store_true', help='use pinned downloads instead of system libraries')
|
||||
ap.add_argument('--no-boost', action='store_true', help='skip Boost.JSON')
|
||||
ap.add_argument('--native', action='store_true', help='compile for this CPU (-march=native / -mcpu=native)')
|
||||
ap.add_argument('--rounds', type=int, default=30, help='rounds of bench_view (default: 30)')
|
||||
ap.add_argument('--corpus', nargs='*', default=[], help='more files for bench_corpus')
|
||||
ap.add_argument('--build-dir', default=os.path.join(HERE, 'build'), help='where to build (default: build/ next to this script)')
|
||||
args = ap.parse_args()
|
||||
|
||||
cxx = os.environ.get('CXX', 'c++')
|
||||
cc = os.environ.get('CC', 'cc')
|
||||
os.makedirs(args.build_dir, exist_ok=True)
|
||||
|
||||
libs = {}
|
||||
for name in ['yyjson', 'simdjson', 'boost']:
|
||||
if name == 'boost' and args.no_boost:
|
||||
continue
|
||||
lib = download_library(name, args.build_dir) if args.download else system_library(name)
|
||||
if lib is None and name != 'boost':
|
||||
sys.exit(f'error: {name} not found; install it, or use --download')
|
||||
if lib is not None:
|
||||
libs[name] = lib
|
||||
with_boost = 'boost' in libs
|
||||
if not with_boost:
|
||||
print('Boost.JSON not found: its columns are skipped', flush=True)
|
||||
|
||||
flags = ['-std=c++17', '-O3', '-DNDEBUG', f'-DJSON_VIEW_BENCH_BOOST={1 if with_boost else 0}']
|
||||
if args.native:
|
||||
flags.append('-mcpu=native' if platform.machine().lower() in ('arm64', 'aarch64') else '-march=native')
|
||||
include = ['-I' + os.path.join(REPO, 'include')] + ['-I' + d for lib in libs.values() for d in lib.include]
|
||||
link = [f for lib in libs.values() for f in lib.link]
|
||||
|
||||
# C sources of downloaded libraries are compiled once
|
||||
objects = []
|
||||
for lib in libs.values():
|
||||
for src in lib.sources:
|
||||
obj = os.path.join(args.build_dir, os.path.basename(src) + '.o')
|
||||
compiler = cc if src.endswith('.c') else cxx
|
||||
run([compiler] + (['-std=c++17'] if compiler == cxx else []) + ['-O3', '-DNDEBUG', '-c', src, '-o', obj]
|
||||
+ ['-I' + d for d in lib.include])
|
||||
objects.append(obj)
|
||||
|
||||
binaries = {}
|
||||
for bench in ['bench_view', 'bench_corpus']:
|
||||
exe = os.path.join(args.build_dir, bench)
|
||||
run([cxx] + flags + include + [os.path.join(HERE, bench + '.cpp')] + objects + link + ['-o', exe])
|
||||
binaries[bench] = exe
|
||||
|
||||
# run: bench_view on its documents, bench_corpus on those and the given files
|
||||
corpus = [os.path.join(args.data, f) for f in DEFAULT_CORPUS] + args.corpus
|
||||
outputs = {}
|
||||
outputs['bench_view'] = run([binaries['bench_view'], args.data, str(args.rounds)], cwd=args.build_dir,
|
||||
capture_output=True, text=True).stdout
|
||||
outputs['bench_corpus'] = run([binaries['bench_corpus']] + corpus, cwd=args.build_dir,
|
||||
capture_output=True, text=True).stdout
|
||||
for name, text in outputs.items():
|
||||
print(text)
|
||||
|
||||
# results with their metadata
|
||||
now = datetime.datetime.now()
|
||||
host = re.sub(r'[^A-Za-z0-9-]+', '-', platform.node().split('.')[0]) or 'host'
|
||||
stem = os.path.join(HERE, 'results', f'{now:%Y-%m-%d}-{host}')
|
||||
os.makedirs(os.path.dirname(stem), exist_ok=True)
|
||||
meta = [
|
||||
('date', f'{now:%Y-%m-%d %H:%M}'),
|
||||
('commit', git_commit()),
|
||||
('CPU', cpu_model()),
|
||||
('OS', f'{platform.system()} {platform.release()} ({platform.machine()})'),
|
||||
('compiler', output([cxx, '--version']).splitlines()[0] if output([cxx, '--version']) else cxx),
|
||||
('flags', ' '.join(flags)),
|
||||
('yyjson', libs['yyjson'].version),
|
||||
('simdjson', libs['simdjson'].version),
|
||||
('Boost.JSON', libs['boost'].version if with_boost else 'skipped (not found)'),
|
||||
('libraries from', 'pinned downloads' if args.download else 'the system'),
|
||||
('rounds', str(args.rounds)),
|
||||
]
|
||||
with open(stem + '.md', 'w', encoding='utf-8') as f:
|
||||
f.write(f'# json_view comparison, {now:%Y-%m-%d}\n\n')
|
||||
f.write('Generated by `tests/benchmarks/json_view/compare.py`; best of the interleaved rounds.\n\n')
|
||||
f.write('| | |\n|---|---|\n')
|
||||
for key, value in meta:
|
||||
f.write(f'| {key} | {value} |\n')
|
||||
for name, text in outputs.items():
|
||||
f.write(f'\n## {name}\n\n```\n{text.rstrip()}\n```\n')
|
||||
with open(stem + '.csv', 'w', encoding='utf-8') as out:
|
||||
out.write(''.join(f'# {key}: {value}\n' for key, value in meta))
|
||||
for name in ['bench_view', 'bench_corpus']:
|
||||
path = os.path.join(args.build_dir, name + '.csv')
|
||||
if os.path.isfile(path):
|
||||
with open(path, encoding='utf-8') as f:
|
||||
out.write(f'# {name}\n' + f.read())
|
||||
print(f'results: {stem}.md, {stem}.csv')
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
@@ -1244,3 +1244,60 @@ TEST_CASE("json_view comparison")
|
||||
CHECK(a.root() != json_document::parse(other).root());
|
||||
}
|
||||
}
|
||||
|
||||
TEST_CASE("json_view large objects")
|
||||
{
|
||||
// objects with 128 members or more are looked up with a hash index
|
||||
for (const std::size_t members :
|
||||
{
|
||||
127u, 128u, 129u, 10000u
|
||||
})
|
||||
{
|
||||
CAPTURE(members);
|
||||
std::string text = "{";
|
||||
for (std::size_t i = 0; i < members; ++i)
|
||||
{
|
||||
text += (i != 0 ? ",\"" : "\"") + std::string(i % 23, 'k') + std::to_string(i) + (i % 7 == 0 ? "\\n" : "") + "\":" + std::to_string(i);
|
||||
}
|
||||
text += R"(,"":"empty key","k1":"a duplicate of an earlier key"})";
|
||||
const json_document d = json_document::parse(text);
|
||||
const json_view v = d.root();
|
||||
const json j = json::parse(text);
|
||||
for (std::size_t i = 0; i < members; ++i)
|
||||
{
|
||||
const std::string key = std::string(i % 23, 'k') + std::to_string(i) + (i % 7 == 0 ? "\n" : "");
|
||||
CHECK(v[key].get<std::size_t>() == i);
|
||||
CHECK(v.contains(key));
|
||||
CHECK(v.find(key).key() == key);
|
||||
CHECK(v.at(key).get<std::size_t>() == i);
|
||||
CHECK(!v.contains(key + "x"));
|
||||
}
|
||||
CHECK(v[""].get_string() == "empty key");
|
||||
CHECK(v["k1"].get<int>() == 1); // the first of duplicate keys, as for small objects
|
||||
CHECK(!v.contains("missing"));
|
||||
CHECK_THROWS_WITH_AS(v.at("missing"), "[json.exception.out_of_range.403] key 'missing' not found", json::out_of_range&);
|
||||
CHECK(v == j);
|
||||
CHECK(v.materialize() == j);
|
||||
}
|
||||
|
||||
SECTION("nested, reused, and in arrays")
|
||||
{
|
||||
std::string inner = "{";
|
||||
for (int i = 0; i < 300; ++i)
|
||||
{
|
||||
inner += (i != 0 ? ",\"m" : "\"m") + std::to_string(i) + "\":" + std::to_string(i);
|
||||
}
|
||||
inner += '}';
|
||||
const std::string text = "[" + inner + ",{\"x\":" + inner + "}," + inner + "]";
|
||||
json_document d = json_document::parse(text);
|
||||
CHECK(d.root()[0]["m299"].get<int>() == 299);
|
||||
CHECK(d.root()[1]["x"]["m150"].get<int>() == 150);
|
||||
CHECK(d.root()[2]["m0"].get<int>() == 0);
|
||||
const std::size_t with_index = d.memory_usage();
|
||||
d.read(std::string("{\"small\": 1}"));
|
||||
CHECK(d.root()["small"].get<int>() == 1);
|
||||
d.read(text);
|
||||
CHECK(d.root()[2]["m7"].get<int>() == 7);
|
||||
CHECK(d.memory_usage() >= with_index / 2);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -17,6 +17,7 @@
|
||||
#endif
|
||||
using nlohmann::json;
|
||||
|
||||
#include <array>
|
||||
#include <cstdint>
|
||||
#include <fstream>
|
||||
#include <memory>
|
||||
@@ -140,6 +141,32 @@ void check_same(const std::string& text)
|
||||
}
|
||||
}
|
||||
|
||||
// a string value and a key must be accepted or rejected as json::parse does,
|
||||
// and give its value (cheaper than check_same: the options do not matter)
|
||||
void check_string(const std::string& content)
|
||||
{
|
||||
for (const std::string& text :
|
||||
{
|
||||
"[\"" + content + "\"]", "{\"" + content + "\":1}"
|
||||
})
|
||||
{
|
||||
const bool accepted = json::accept(text);
|
||||
for (const bool sentinel :
|
||||
{
|
||||
true, false
|
||||
})
|
||||
{
|
||||
const built b = build(text, false, false, sentinel);
|
||||
if (b.ok != accepted || (accepted && value_of(b) != json::parse(text)))
|
||||
{
|
||||
CAPTURE(text);
|
||||
CHECK(b.ok == accepted);
|
||||
CHECK((b.ok && accepted ? value_of(b) == json::parse(text) : true));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// a small deterministic generator of documents
|
||||
struct generator
|
||||
{
|
||||
@@ -368,3 +395,81 @@ TEST_CASE("json_view builder")
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
TEST_CASE("json_view builder: strings across vector blocks")
|
||||
{
|
||||
// Strings are scanned 8 or 16 bytes at a time (NEON, SSE2, or SWAR) and
|
||||
// non-ASCII text with the vector UTF-8 check (NEON, SSSE3) or one sequence
|
||||
// at a time. Sequences are placed so that they start at every offset
|
||||
// around the block boundaries of keys (16, 32) and values (8, 24), with
|
||||
// text of several lengths after them.
|
||||
const std::array<std::size_t, 16> prefixes = {{0, 6, 7, 8, 13, 14, 15, 16, 21, 22, 23, 24, 29, 30, 31, 32}};
|
||||
const std::array<std::size_t, 3> suffixes = {{0, 3, 17}};
|
||||
const auto around = [&](const std::string & seq, std::size_t prefix, std::size_t suffix)
|
||||
{
|
||||
return std::string(prefix, 'a') + seq + std::string(suffix, 'b');
|
||||
};
|
||||
|
||||
SECTION("every two-byte sequence")
|
||||
{
|
||||
for (unsigned lead = 0x80; lead <= 0xFF; ++lead)
|
||||
{
|
||||
for (unsigned second = 0; second <= 0xFF; ++second)
|
||||
{
|
||||
if (second == '"' || second == '\\')
|
||||
{
|
||||
continue;
|
||||
}
|
||||
const std::string seq = {static_cast<char>(lead), static_cast<char>(second)};
|
||||
check_string(around(seq, prefixes[(lead + second) % 16], suffixes[second % 3]));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("three- and four-byte sequences")
|
||||
{
|
||||
const std::array<unsigned, 8> conts = {{0x7F, 0x80, 0x8F, 0x90, 0x9F, 0xA0, 0xBF, 0xC0}};
|
||||
for (unsigned lead = 0xE0; lead <= 0xF7; ++lead)
|
||||
{
|
||||
for (const unsigned b2 : conts)
|
||||
{
|
||||
for (const unsigned b3 : conts)
|
||||
{
|
||||
for (const std::size_t prefix : prefixes)
|
||||
{
|
||||
std::string seq = {static_cast<char>(lead), static_cast<char>(b2), static_cast<char>(b3)};
|
||||
if (lead >= 0xF0)
|
||||
{
|
||||
seq += static_cast<char>(prefix % 2 == 0 ? 0x80 : 0xBF);
|
||||
}
|
||||
check_string(around(seq, prefix, suffixes[prefix % 3]));
|
||||
// cut short before the end of the string
|
||||
check_string(around(seq.substr(0, seq.size() - 1), prefix, suffixes[prefix % 3]));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("long runs of text with one damaged byte")
|
||||
{
|
||||
const std::array<const char*, 7> chars = {{"a", "\xc3\xa9", "\xe3\x81\x82", "\xf0\x9f\x98\x80", "\xed\x9f\xbf", "\xef\xbf\xbf", "\xf4\x8f\xbf\xbf"}};
|
||||
const std::array<char, 11> damage = {{'\x80', '\xbf', '\xc0', '\xc1', '\xe0', '\xed', '\xf5', '\xff', '\x1f', '"', '\\'}};
|
||||
std::mt19937 rng(5295); // NOLINT(cert-msc32-c,cert-msc51-cpp,bugprone-random-generator-seed): reproducible
|
||||
for (int i = 0; i < 4000; ++i)
|
||||
{
|
||||
std::string text;
|
||||
const auto n = rng() % 60;
|
||||
for (unsigned k = 0; k < n; ++k)
|
||||
{
|
||||
text += chars[rng() % chars.size()];
|
||||
}
|
||||
check_string(text);
|
||||
if (!text.empty())
|
||||
{
|
||||
text[rng() % text.size()] = damage[rng() % damage.size()];
|
||||
check_string(text);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user