mirror of
https://github.com/nlohmann/json.git
synced 2026-09-30 19:50:34 +00:00
Compare commits
7
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
eede67ca92 | ||
|
|
82b31f31b8 | ||
|
|
798f4c888a | ||
|
|
98258aa3af | ||
|
|
904c8c6710 | ||
|
|
1f0c3be6f3 | ||
|
|
c3b51addbf |
@@ -78,9 +78,11 @@ cc_library(
|
||||
"include/nlohmann/detail/view/materialize.hpp",
|
||||
"include/nlohmann/detail/view/node.hpp",
|
||||
"include/nlohmann/detail/view/number.hpp",
|
||||
"include/nlohmann/detail/view/object_index.hpp",
|
||||
"include/nlohmann/detail/view/pointer.hpp",
|
||||
"include/nlohmann/detail/view/scan.hpp",
|
||||
"include/nlohmann/detail/view/serializer.hpp",
|
||||
"include/nlohmann/detail/view/simd.hpp",
|
||||
"include/nlohmann/detail/view/string_ref.hpp",
|
||||
"include/nlohmann/detail/view/value.hpp",
|
||||
"include/nlohmann/json.hpp",
|
||||
|
||||
@@ -1405,6 +1405,7 @@ THE SOFTWARE IS PROVIDED “AS IS”, WITHOUT WARRANTY OF ANY KIND, EXPRESS OR I
|
||||
- The class contains parts of [Google Abseil](https://github.com/abseil/abseil-cpp) which is licensed under the [Apache 2.0 License](https://opensource.org/licenses/Apache-2.0).
|
||||
- The class contains an adapted version of the Eisel-Lemire algorithm, its table of powers of five, and its digit comparison for long numbers from [fast_float](https://github.com/fastfloat/fast_float) by Daniel Lemire and contributors, which is available under the [MIT License](https://opensource.org/licenses/MIT) (used here), the Apache 2.0 License, and the Boost Software License. Copyright © 2021 The fast_float authors
|
||||
- The view's parser (`<nlohmann/json_view.hpp>`) contains techniques and code adapted from [yyjson](https://github.com/ibireme/yyjson) by YaoYuan, which is licensed under the [MIT License](https://opensource.org/licenses/MIT) (see above): table-driven decoding of `\u` escapes and fixed-offset unrolled checks.
|
||||
- The view's parser (`<nlohmann/json_view.hpp>`) validates non-ASCII strings with the vector UTF-8 check of [simdjson](https://github.com/simdjson/simdjson) by Daniel Lemire, Geoff Langdale, John Keiser, and contributors (its "lookup4" algorithm and tables, after J. Keiser and D. Lemire, "Validating UTF-8 In Less Than One Instruction Per Byte", 2021), which is available under the [MIT License](https://opensource.org/licenses/MIT) (used here) and the Apache 2.0 License. Copyright © 2018-2025 The simdjson authors
|
||||
|
||||
<img align="right" src="https://git.fsfe.org/reuse/reuse-ci/raw/branch/master/reuse-horizontal.png" alt="REUSE Software">
|
||||
|
||||
|
||||
@@ -307,3 +307,5 @@ INSERT INTO searchIndex(name, type, path) VALUES ('NLOHMANN_JSON_SERIALIZE_ENUM'
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('NLOHMANN_JSON_VERSION_MAJOR', 'Macro', 'api/macros/nlohmann_json_version_major/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('NLOHMANN_JSON_VERSION_MINOR', 'Macro', 'api/macros/nlohmann_json_version_major/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('NLOHMANN_JSON_VERSION_PATCH', 'Macro', 'api/macros/nlohmann_json_version_major/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('JSON_VIEW_NO_SIMD', 'Macro', 'api/macros/json_view_no_simd/index.html');
|
||||
INSERT INTO searchIndex(name, type, path) VALUES ('JSON_VIEW_USE_SSSE3', 'Macro', 'api/macros/json_view_use_ssse3/index.html');
|
||||
|
||||
@@ -74,6 +74,8 @@ None of these exceptions carry a [`JSON_DIAGNOSTICS`](../macros/json_diagnostics
|
||||
1. Linear in the number of members: as for [`ordered_json`](../ordered_json.md), members are compared one after
|
||||
another, in document order, stopping at the first match. Each comparison first checks the key's length --
|
||||
already known from the index, without reading the key bytes -- before comparing its content.
|
||||
Objects with 128 or more members get a hash index while parsing, so that a lookup in them takes constant time
|
||||
on average.
|
||||
2. Linear in `idx`: elements are skipped one at a time from the first one, since they are not a fixed size in the
|
||||
index (unlike `BasicJsonType`'s array, which is random-access).
|
||||
3. Linear in the number of reference tokens of `ptr` and, for each token, in the number of members of the object at
|
||||
|
||||
@@ -35,6 +35,8 @@ No-throw guarantee: this function never throws exceptions.
|
||||
1. Linear in the number of members: as for [`ordered_json`](../ordered_json.md), members are compared one after
|
||||
another, in document order, stopping at the first match. Each comparison first checks the key's length -- already
|
||||
known from the index, without reading the key bytes -- before comparing its content.
|
||||
Objects with 128 or more members get a hash index while parsing, so that a lookup in them takes constant time
|
||||
on average.
|
||||
2. Linear in the number of reference tokens of `ptr` and, for each token, in the number of members of the object at
|
||||
that level or the index into the array -- as for [`operator[]`](operator[].md#complexity) and
|
||||
[`at`](at.md#complexity) with a JSON pointer.
|
||||
|
||||
@@ -26,6 +26,8 @@ No-throw guarantee: this function never throws exceptions.
|
||||
Linear in the number of members: as for [`ordered_json`](../ordered_json.md), members are compared one after
|
||||
another, in document order, stopping at the first match. Each comparison first checks the key's length -- already
|
||||
known from the index, without reading the key bytes -- before comparing its content.
|
||||
Objects with 128 or more members get a hash index while parsing, so that a lookup in them takes constant time on
|
||||
average.
|
||||
|
||||
## Notes
|
||||
|
||||
|
||||
@@ -28,6 +28,8 @@ No-throw guarantee: this function never throws exceptions.
|
||||
Linear in the number of members: as for [`ordered_json`](../ordered_json.md), members are compared one after
|
||||
another, in document order, stopping at the first match. Each comparison first checks the key's length -- already
|
||||
known from the index, without reading the key bytes -- before comparing its content.
|
||||
Objects with 128 or more members get a hash index while parsing, so that a lookup in them takes constant time on
|
||||
average.
|
||||
|
||||
## Notes
|
||||
|
||||
|
||||
@@ -76,6 +76,8 @@ None of these exceptions carry a [`JSON_DIAGNOSTICS`](../macros/json_diagnostics
|
||||
another, in document order, stopping at the first match. Each comparison first checks the key's length --
|
||||
already known from the index, without reading the key bytes -- before comparing its content, so a key of a
|
||||
different length than `key` is rejected without touching the source text.
|
||||
Objects with 128 or more members get a hash index while parsing, so that a lookup in them takes constant time
|
||||
on average.
|
||||
2. Linear in `idx`: elements are skipped one at a time from the first one, since they are not a fixed size in the
|
||||
index (unlike `BasicJsonType`'s array, which is random-access).
|
||||
3. Linear in the number of reference tokens of `ptr` and, for each token, in the number of members of the object at
|
||||
|
||||
@@ -70,6 +70,8 @@ None of these exceptions carry a [`JSON_DIAGNOSTICS`](../macros/json_diagnostics
|
||||
1. Linear in the number of members: as for [`operator[]`](operator[].md#complexity), members are compared one after
|
||||
another, in document order, stopping at the first match. Plus the complexity of converting the found member to
|
||||
`T` (see [`get`](get.md)).
|
||||
Objects with 128 or more members get a hash index while parsing, so that a lookup in them takes constant time
|
||||
on average.
|
||||
2. Linear in the number of reference tokens of `ptr` and, for each token, in the number of members of the object at
|
||||
that level or the index into the array -- as for the [`operator[]`](operator[].md#complexity) and
|
||||
[`at`](at.md#complexity) overloads that take a JSON pointer. Plus the complexity of converting the resolved value
|
||||
|
||||
@@ -34,6 +34,8 @@ header. See also the [macro overview page](../../features/macros.md).
|
||||
- [**JSON_SKIP_UNSUPPORTED_COMPILER_CHECK**](json_skip_unsupported_compiler_check.md) - do not warn about unsupported compilers
|
||||
- [**JSON_USE_GLOBAL_UDLS**](json_use_global_udls.md) - place user-defined string literals (UDLs) into the global namespace
|
||||
- [**JSON_USE_SIMDUTF**](json_use_simdutf.md) - use the simdutf library to accelerate UTF-8 validation
|
||||
- [**JSON_VIEW_NO_SIMD**](json_view_no_simd.md) - use only portable code in the parser of `json_view.hpp`
|
||||
- [**JSON_VIEW_USE_SSSE3**](json_view_use_ssse3.md) - validate non-ASCII strings with SSSE3 in the parser of `json_view.hpp`
|
||||
|
||||
## Library version
|
||||
|
||||
|
||||
@@ -0,0 +1,50 @@
|
||||
# JSON_VIEW_NO_SIMD
|
||||
|
||||
```cpp
|
||||
#define JSON_VIEW_NO_SIMD
|
||||
```
|
||||
|
||||
When defined, the parser of [`basic_json_document`](../basic_json_document/index.md) (`<nlohmann/json_view.hpp>`)
|
||||
uses only portable C++ to scan strings. By default, it scans long runs of string bytes 16 at a time with NEON on
|
||||
AArch64 (with GCC and Clang) and SSE2 on x86-64, which are part of the baseline instruction sets of these
|
||||
architectures, and validates non-ASCII text with NEON (or SSSE3, see
|
||||
[`JSON_VIEW_USE_SSSE3`](json_view_use_ssse3.md)).
|
||||
|
||||
The same input is accepted or rejected either way, with the same values, and errors are reported the same way; only
|
||||
the speed differs. The macro exists for platforms whose compilers lack the intrinsics headers, and to test the portable
|
||||
code.
|
||||
|
||||
!!! warning "Define consistently"
|
||||
|
||||
The macro selects between two definitions of the same inline functions. It must therefore be defined identically for
|
||||
**every** translation unit that includes `<nlohmann/json_view.hpp>`; prefer a compile definition on the target.
|
||||
|
||||
## Default definition
|
||||
|
||||
By default, `#!cpp JSON_VIEW_NO_SIMD` is not defined, and the vector code is used where available.
|
||||
|
||||
```cpp
|
||||
#undef JSON_VIEW_NO_SIMD
|
||||
```
|
||||
|
||||
## Examples
|
||||
|
||||
??? example
|
||||
|
||||
The code below uses the portable string scanning of the view.
|
||||
|
||||
```cpp
|
||||
#define JSON_VIEW_NO_SIMD
|
||||
#include <nlohmann/json_view.hpp>
|
||||
|
||||
...
|
||||
```
|
||||
|
||||
## See also
|
||||
|
||||
- [JSON_VIEW_USE_SSSE3](json_view_use_ssse3.md) - validate non-ASCII strings with SSSE3 on x86-64
|
||||
- [json_view](../../features/json_view.md) - the zero-copy view
|
||||
|
||||
## Version history
|
||||
|
||||
- Added in version 3.13.0.
|
||||
@@ -0,0 +1,49 @@
|
||||
# JSON_VIEW_USE_SSSE3
|
||||
|
||||
```cpp
|
||||
#define JSON_VIEW_USE_SSSE3
|
||||
```
|
||||
|
||||
When defined on x86-64, the parser of [`basic_json_document`](../basic_json_document/index.md)
|
||||
(`<nlohmann/json_view.hpp>`) validates non-ASCII text in strings with SSSE3, 16 bytes at a time, using the "lookup4"
|
||||
algorithm of [simdjson](https://github.com/simdjson/simdjson). Without it, non-ASCII text is validated one UTF-8
|
||||
sequence at a time on x86-64; on AArch64, the vector check uses NEON and is always on.
|
||||
|
||||
SSSE3 is not part of the x86-64 baseline, so the code must be compiled for it: define the macro only together with a
|
||||
compiler option that enables SSSE3 (e.g. `-mssse3`, or `-march=` with a CPU that has it), and only for programs that
|
||||
run on such CPUs. The same input is accepted or rejected either way; only the speed of non-ASCII text differs.
|
||||
|
||||
!!! warning "Define consistently"
|
||||
|
||||
The macro selects between two definitions of the same inline functions. It must therefore be defined identically,
|
||||
with the same compiler options, for **every** translation unit that includes `<nlohmann/json_view.hpp>`; mixing
|
||||
translation units that define it with ones that do not is an ODR violation. Prefer a compile definition on the
|
||||
target.
|
||||
|
||||
## Default definition
|
||||
|
||||
By default, `#!cpp JSON_VIEW_USE_SSSE3` is not defined.
|
||||
|
||||
```cpp
|
||||
#undef JSON_VIEW_USE_SSSE3
|
||||
```
|
||||
|
||||
## Examples
|
||||
|
||||
??? example
|
||||
|
||||
With CMake, for a program that only runs on CPUs with SSSE3:
|
||||
|
||||
```cmake
|
||||
target_compile_definitions(your_target PRIVATE JSON_VIEW_USE_SSSE3)
|
||||
target_compile_options(your_target PRIVATE -mssse3)
|
||||
```
|
||||
|
||||
## See also
|
||||
|
||||
- [JSON_VIEW_NO_SIMD](json_view_no_simd.md) - use only portable code in the view's parser
|
||||
- [JSON_USE_SIMDUTF](json_use_simdutf.md) - validate UTF-8 with simdutf in `basic_json`'s parser
|
||||
|
||||
## Version history
|
||||
|
||||
- Added in version 3.13.0.
|
||||
@@ -196,7 +196,7 @@ packet-beta
|
||||
|-------|---------|------------|-------------------------------------------------------------------------------------------------------------------------------|
|
||||
| 0 | `kind` | `uint8_t` | the type, numbered as [`value_t`](../api/basic_json/value_t.md): 0 null, 1 object, 2 array, 3 string, 4 boolean, 5 signed integer, 6 unsigned integer, 7 float |
|
||||
| 1 | `flags` | `uint8_t` | bits 0-1: where a string's bytes are (0: the source text, 1: the buffer of decoded strings, for strings with escapes); bit 2: the value of a boolean |
|
||||
| 2-3 | `extra` | `uint16_t` | numbers: the number of integer digits (low byte) and fraction digits (high byte), 255 for more; otherwise 0 |
|
||||
| 2-3 | `extra` | `uint16_t` | numbers: the number of integer digits (low byte) and fraction digits (high byte), 255 for more; objects: the number of their hash index (1-based), or 0; otherwise 0 |
|
||||
| 4-7 | `off` | `uint32_t` | where the value starts: the first byte after a string's opening quote (or its position in the buffer of decoded strings), the first byte of a number or literal, the bracket of an array or object |
|
||||
| 8-11 | `len` | `uint32_t` | strings: the length after decoding; floats and literals: the length of the token; arrays and objects: the number of elements |
|
||||
| 12-15 | `next` | `uint32_t` | arrays and objects: the number of nodes of the subtree, including the node itself |
|
||||
@@ -211,6 +211,11 @@ packet-beta
|
||||
subtree is `next` nodes further for an array or object, and the next node otherwise (`document_data::after`). Views
|
||||
step from element to element this way and skip whole subtrees in constant time.
|
||||
- **Offsets** are 32 bits wide, so a document is limited to 4 GiB (`out_of_range.416`).
|
||||
- **Large objects** (128 members or more) get a hash index after parsing
|
||||
([`detail/view/object_index.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/view/object_index.hpp)):
|
||||
an open-addressing table whose slots hold the distance from the object's node to a key's node, so that a lookup does
|
||||
not compare every key. The object's `extra` holds the number of its table. Only 65,535 tables fit into `extra`;
|
||||
objects beyond them are searched linearly.
|
||||
|
||||
For example, `#!json {"a": [1, 2.5]}` becomes five nodes. Each node's elements follow it, and `next` leads from an
|
||||
array or object past its subtree:
|
||||
|
||||
@@ -23,3 +23,5 @@ The class contains a copy of [Hedley](https://nemequ.github.io/hedley/) from Eva
|
||||
The class contains an adapted version of the Eisel-Lemire algorithm, its table of powers of five, and its digit comparison for long numbers from [fast_float](https://github.com/fastfloat/fast_float) by Daniel Lemire and contributors, which is available under the [MIT License](https://opensource.org/licenses/MIT) (used here), the Apache 2.0 License, and the Boost Software License. Copyright © 2021 The fast_float authors
|
||||
|
||||
The view's parser (`<nlohmann/json_view.hpp>`) contains techniques and code adapted from [yyjson](https://github.com/ibireme/yyjson) by YaoYuan, which is licensed under the [MIT License](https://opensource.org/licenses/MIT) (see above): table-driven decoding of `\u` escapes and fixed-offset unrolled checks.
|
||||
|
||||
The view's parser (`<nlohmann/json_view.hpp>`) validates non-ASCII strings with the vector UTF-8 check of [simdjson](https://github.com/simdjson/simdjson) by Daniel Lemire, Geoff Langdale, John Keiser, and contributors (its "lookup4" algorithm and tables, after J. Keiser and D. Lemire, "Validating UTF-8 In Less Than One Instruction Per Byte", 2021), which is available under the [MIT License](https://opensource.org/licenses/MIT) (used here) and the Apache 2.0 License. Copyright © 2018-2025 The simdjson authors
|
||||
|
||||
@@ -369,6 +369,8 @@ nav:
|
||||
- 'JSON_USE_IMPLICIT_CONVERSIONS': api/macros/json_use_implicit_conversions.md
|
||||
- 'JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON': api/macros/json_use_legacy_discarded_value_comparison.md
|
||||
- 'JSON_USE_SIMDUTF': api/macros/json_use_simdutf.md
|
||||
- 'JSON_VIEW_NO_SIMD': api/macros/json_view_no_simd.md
|
||||
- 'JSON_VIEW_USE_SSSE3': api/macros/json_view_use_ssse3.md
|
||||
- 'NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE, NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE_WITH_DEFAULT, NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE_ONLY_SERIALIZE, NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE, NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_WITH_DEFAULT, NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_ONLY_SERIALIZE': api/macros/nlohmann_define_derived_type.md
|
||||
- 'NLOHMANN_DEFINE_TYPE_INTRUSIVE, NLOHMANN_DEFINE_TYPE_INTRUSIVE_WITH_DEFAULT, NLOHMANN_DEFINE_TYPE_INTRUSIVE_ONLY_SERIALIZE': api/macros/nlohmann_define_type_intrusive.md
|
||||
- 'NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE, NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE_WITH_DEFAULT, NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE_ONLY_SERIALIZE': api/macros/nlohmann_define_type_non_intrusive.md
|
||||
|
||||
@@ -116,6 +116,13 @@ class builder
|
||||
frame shallow[64]; // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays): not initialized on purpose; filled as containers open
|
||||
std::vector<frame> deep{};
|
||||
|
||||
/// remember an object to index after parsing (out of line, so that the
|
||||
/// parse loop only has a call for it)
|
||||
NLOHMANN_VIEW_NOINLINE void note_large_object(std::uint32_t idx)
|
||||
{
|
||||
doc.large_objects.push_back(idx);
|
||||
}
|
||||
|
||||
NLOHMANN_VIEW_NOINLINE bool fail(error_code c, const unsigned char* at) noexcept
|
||||
{
|
||||
m_failure.code = c;
|
||||
@@ -503,7 +510,7 @@ class builder
|
||||
switch (cur()) \
|
||||
{ \
|
||||
case '"': \
|
||||
if (NLOHMANN_VIEW_UNLIKELY(!string())) { return false; } \
|
||||
if (NLOHMANN_VIEW_UNLIKELY(!string<true>())) { return false; } \
|
||||
goto NEXT; \
|
||||
case '{': \
|
||||
open(value_t::object); \
|
||||
@@ -587,7 +594,7 @@ obj_key:
|
||||
{
|
||||
return fail(error_code::expected_key);
|
||||
}
|
||||
if (NLOHMANN_VIEW_UNLIKELY(!string()))
|
||||
if (NLOHMANN_VIEW_UNLIKELY(!string<false>()))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
@@ -628,19 +635,27 @@ obj_next:
|
||||
if (enabled(TrailingCommas) && cur() == '}')
|
||||
{
|
||||
++p;
|
||||
goto close_container;
|
||||
goto close_object;
|
||||
}
|
||||
goto obj_key;
|
||||
}
|
||||
if (cur() == '}')
|
||||
{
|
||||
++p;
|
||||
goto close_container;
|
||||
goto close_object;
|
||||
}
|
||||
return fail(error_code::expected_object_end);
|
||||
|
||||
#undef NLOHMANN_VIEW_VALUE
|
||||
|
||||
close_object:
|
||||
// a large object gets a hash index (objects only, so that closing
|
||||
// an array pays nothing for this)
|
||||
if (NLOHMANN_VIEW_UNLIKELY(cur_count >= document_data::index_min_members))
|
||||
{
|
||||
cold.note_large_object(cur_idx);
|
||||
}
|
||||
|
||||
close_container:
|
||||
close();
|
||||
if (NLOHMANN_VIEW_UNLIKELY(depth == 0))
|
||||
@@ -690,7 +705,7 @@ root_done:
|
||||
switch (cur())
|
||||
{
|
||||
case '"':
|
||||
return string();
|
||||
return string<true>();
|
||||
case 't':
|
||||
return literal("true", 4, value_t::boolean, node_flags::is_true);
|
||||
case 'f':
|
||||
@@ -977,12 +992,13 @@ indent_done:
|
||||
return true;
|
||||
}
|
||||
|
||||
/// a string (value or key) at p
|
||||
/// a string at p: a value (Value) or a key
|
||||
template<bool Value>
|
||||
NLOHMANN_VIEW_ALWAYS_INLINE bool string()
|
||||
{
|
||||
++p; // opening quote
|
||||
const unsigned char* const s = p;
|
||||
p = scan_string_run(p, e);
|
||||
p = scan_string_run<Value>(p, e);
|
||||
if (NLOHMANN_VIEW_LIKELY(p != e && *p == '"'))
|
||||
{
|
||||
emit(value_t::string, 0, 0, static_cast<std::size_t>(s - b), static_cast<std::uint64_t>(p - s));
|
||||
|
||||
@@ -10,9 +10,11 @@
|
||||
|
||||
#include <array> // array
|
||||
#include <cstddef> // size_t
|
||||
#include <cstdint> // uint32_t
|
||||
#include <cstring> // memcpy
|
||||
#include <new> // operator new, placement new
|
||||
#include <string> // string
|
||||
#include <vector> // vector
|
||||
|
||||
#include <nlohmann/json.hpp>
|
||||
#include <nlohmann/detail/view/macro_scope.hpp>
|
||||
@@ -37,6 +39,17 @@ struct document_data
|
||||
std::size_t inline_cap = 0;
|
||||
std::string arena{}; ///< decoded strings that contained escapes // NOLINT(readability-redundant-member-init)
|
||||
std::string owned{}; ///< owned copy of the input, if any // NOLINT(readability-redundant-member-init)
|
||||
|
||||
// hash indexes of large objects (see object_index.hpp)
|
||||
static constexpr std::uint32_t index_min_members = 128;
|
||||
struct object_index
|
||||
{
|
||||
std::size_t start; ///< first slot in index_slots
|
||||
std::uint32_t mask; ///< slot count - 1 (a power of two minus one)
|
||||
};
|
||||
std::vector<object_index> indexes{}; // NOLINT(readability-redundant-member-init)
|
||||
std::vector<std::uint32_t> index_slots{}; // NOLINT(readability-redundant-member-init)
|
||||
std::vector<std::uint32_t> large_objects{}; ///< positions of the objects to index (noted while parsing) // NOLINT(readability-redundant-member-init)
|
||||
std::array<const char*, 4> base = {{nullptr, nullptr, nullptr, nullptr}}; ///< string bases: source, arena (indexed by flags & node_flags::storage)
|
||||
bool discarded = true;
|
||||
|
||||
@@ -64,7 +77,7 @@ struct document_data
|
||||
}
|
||||
};
|
||||
|
||||
document_data() noexcept = default;
|
||||
document_data() = default;
|
||||
document_data(const document_data&) = delete;
|
||||
document_data(document_data&&) = delete;
|
||||
document_data& operator=(const document_data&) = delete;
|
||||
|
||||
@@ -16,6 +16,7 @@
|
||||
#include <nlohmann/detail/view/document_data.hpp>
|
||||
#include <nlohmann/detail/view/macro_scope.hpp>
|
||||
#include <nlohmann/detail/view/node.hpp>
|
||||
#include <nlohmann/detail/view/object_index.hpp>
|
||||
|
||||
NLOHMANN_JSON_NAMESPACE_BEGIN
|
||||
namespace detail
|
||||
@@ -85,6 +86,10 @@ class short_key
|
||||
/// nullptr; most keys are rejected by their length, from the index alone
|
||||
inline const node* find_member(const document_data& d, const node* object, const char* key, std::size_t n) noexcept
|
||||
{
|
||||
if (NLOHMANN_VIEW_UNLIKELY(object->extra != 0))
|
||||
{
|
||||
return find_indexed(d, object, key, n); // a large object
|
||||
}
|
||||
const node* const end = document_data::child_end(object);
|
||||
const auto* const k = reinterpret_cast<const unsigned char*>(key); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
|
||||
if (NLOHMANN_VIEW_LIKELY(n <= 16))
|
||||
|
||||
@@ -19,3 +19,8 @@
|
||||
#undef NLOHMANN_VIEW_THROW
|
||||
#undef NLOHMANN_VIEW_LITTLE_ENDIAN
|
||||
#undef NLOHMANN_VIEW_REPEAT16
|
||||
#undef NLOHMANN_VIEW_NEON
|
||||
#undef NLOHMANN_VIEW_SSE2
|
||||
#undef NLOHMANN_VIEW_SSSE3
|
||||
#undef NLOHMANN_VIEW_VECTOR
|
||||
#undef NLOHMANN_VIEW_VECTOR_UTF8
|
||||
|
||||
@@ -0,0 +1,131 @@
|
||||
// __ _____ _____ _____
|
||||
// __| | __| | | | JSON for Modern C++
|
||||
// | | |__ | | | | | | version 3.12.0
|
||||
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
|
||||
//
|
||||
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <cstddef> // size_t
|
||||
#include <cstdint> // uint32_t, uint64_t
|
||||
#include <cstring> // memcmp
|
||||
|
||||
#include <nlohmann/json.hpp>
|
||||
#include <nlohmann/detail/view/document_data.hpp>
|
||||
#include <nlohmann/detail/view/macro_scope.hpp>
|
||||
#include <nlohmann/detail/view/node.hpp>
|
||||
|
||||
// Hash indexes of large objects, so that a lookup does not compare thousands
|
||||
// of keys (as Boost.JSON switches from a linear search to a hash table for
|
||||
// large objects). An object with document_data::index_min_members members or
|
||||
// more gets an open-addressing table after parsing; its node stores the
|
||||
// number of the table (1-based) in `extra`. A slot holds the offset of a key
|
||||
// node from its object node (0: empty). Of duplicate keys, the first is kept,
|
||||
// as for the linear search.
|
||||
|
||||
NLOHMANN_JSON_NAMESPACE_BEGIN
|
||||
namespace detail
|
||||
{
|
||||
namespace view
|
||||
{
|
||||
|
||||
/// hash of a key: its bytes, eight at a time, in a fixed byte order
|
||||
inline std::uint64_t key_hash(const char* s, std::size_t n) noexcept
|
||||
{
|
||||
std::uint64_t h = 0x9E3779B97F4A7C15u * (n + 1);
|
||||
const auto* p = reinterpret_cast<const unsigned char*>(s); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
|
||||
while (n >= 8)
|
||||
{
|
||||
h = (h ^ read_eight_bytes(p)) * 0xBF58476D1CE4E5B9u;
|
||||
h ^= h >> 29u;
|
||||
p += 8;
|
||||
n -= 8;
|
||||
}
|
||||
std::uint64_t w = 0;
|
||||
for (std::size_t i = 0; i < n; ++i)
|
||||
{
|
||||
w |= static_cast<std::uint64_t>(p[i]) << (8u * i);
|
||||
}
|
||||
h = (h ^ w) * 0x94D049BB133111EBu;
|
||||
return h ^ (h >> 31u);
|
||||
}
|
||||
|
||||
/// build the table of a large object
|
||||
inline void build_object_index(document_data& d, node* obj)
|
||||
{
|
||||
if (d.indexes.size() >= 0xFFFFu)
|
||||
{
|
||||
return; // LCOV_EXCL_LINE (the number must fit `extra`; more large objects are searched linearly)
|
||||
}
|
||||
std::size_t cap = 16;
|
||||
while (cap < 2 * static_cast<std::size_t>(obj->len))
|
||||
{
|
||||
cap *= 2;
|
||||
}
|
||||
const std::size_t start = d.index_slots.size();
|
||||
d.index_slots.resize(start + cap, 0);
|
||||
std::uint32_t* const slots = d.index_slots.data() + start;
|
||||
const std::size_t mask = cap - 1;
|
||||
for (const node* k = document_data::first_child(obj), *end = document_data::child_end(obj); k != end; k = document_data::after(k + 1))
|
||||
{
|
||||
const char* const key = d.str(*k);
|
||||
const std::uint64_t hash = key_hash(key, k->len); // (a cast of the call would be useless where std::uint64_t is std::size_t)
|
||||
std::size_t i = static_cast<std::size_t>(hash) & mask;
|
||||
bool duplicate = false;
|
||||
while (slots[i] != 0)
|
||||
{
|
||||
const node* const other = obj + slots[i];
|
||||
if (other->len == k->len && (k->len == 0 || std::memcmp(d.str(*other), key, k->len) == 0))
|
||||
{
|
||||
duplicate = true; // keep the first
|
||||
break;
|
||||
}
|
||||
i = (i + 1) & mask;
|
||||
}
|
||||
if (!duplicate)
|
||||
{
|
||||
slots[i] = static_cast<std::uint32_t>(k - obj);
|
||||
}
|
||||
}
|
||||
d.indexes.push_back(document_data::object_index{start, static_cast<std::uint32_t>(mask)});
|
||||
obj->extra = static_cast<std::uint16_t>(d.indexes.size());
|
||||
}
|
||||
|
||||
/// build the tables of the large objects the parser noted
|
||||
inline void build_object_indexes(document_data& d)
|
||||
{
|
||||
for (const std::uint32_t i : d.large_objects)
|
||||
{
|
||||
build_object_index(d, d.tape + i);
|
||||
}
|
||||
}
|
||||
|
||||
/// the key node of the first member with this key of an indexed object, or
|
||||
/// nullptr
|
||||
inline const node* find_indexed(const document_data& d, const node* obj, const char* key, std::size_t n) noexcept
|
||||
{
|
||||
const document_data::object_index& ix = d.indexes[obj->extra - 1u];
|
||||
const std::uint32_t* const slots = d.index_slots.data() + ix.start;
|
||||
const std::uint64_t hash = key_hash(key, n); // (a cast of the call would be useless where std::uint64_t is std::size_t)
|
||||
std::size_t i = static_cast<std::size_t>(hash) & ix.mask;
|
||||
for (;;)
|
||||
{
|
||||
const std::uint32_t s = slots[i];
|
||||
if (s == 0)
|
||||
{
|
||||
return nullptr;
|
||||
}
|
||||
const node* const k = obj + s;
|
||||
if (k->len == n && (n == 0 || std::memcmp(d.str(*k), key, n) == 0))
|
||||
{
|
||||
return k;
|
||||
}
|
||||
i = (i + 1) & ix.mask;
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace view
|
||||
} // namespace detail
|
||||
NLOHMANN_JSON_NAMESPACE_END
|
||||
@@ -16,6 +16,7 @@
|
||||
|
||||
#include <nlohmann/json.hpp>
|
||||
#include <nlohmann/detail/view/macro_scope.hpp>
|
||||
#include <nlohmann/detail/view/simd.hpp>
|
||||
|
||||
// Scanning primitives of the view's parser. The unrolled checks at fixed
|
||||
// offsets follow yyjson (https://github.com/ibireme/yyjson, MIT license): the
|
||||
@@ -62,9 +63,12 @@ NLOHMANN_VIEW_ALWAYS_INLINE std::uint16_t load16(const unsigned char* p) noexcep
|
||||
|
||||
/// Advance over plain string bytes and well-formed UTF-8. Stops at a quote,
|
||||
/// a backslash, a control character, ill-formed UTF-8, or the end. The first
|
||||
/// 16 bytes are checked one by one, so that the position advances by
|
||||
/// constants in predicted branches (most strings are short); longer runs
|
||||
/// continue eight bytes at a time.
|
||||
/// bytes are checked one by one, so that the position advances by constants
|
||||
/// in predicted branches: 16 for keys, whose lengths repeat from record to
|
||||
/// record, and 8 for string values (Value) where a vector loop follows, as
|
||||
/// their lengths vary more. Longer runs continue 16 bytes at a time with NEON
|
||||
/// or SSE2, else eight bytes at a time.
|
||||
template<bool Value = false>
|
||||
NLOHMANN_VIEW_ALWAYS_INLINE const unsigned char* scan_string_run(const unsigned char* p, const unsigned char* e) noexcept
|
||||
{
|
||||
const std::uint8_t* plain = string_plain();
|
||||
@@ -73,9 +77,23 @@ NLOHMANN_VIEW_ALWAYS_INLINE const unsigned char* scan_string_run(const unsigned
|
||||
if (e - p >= 16)
|
||||
{
|
||||
#define NLOHMANN_VIEW_STEP(i) if (NLOHMANN_VIEW_LIKELY(plain[p[i]] != 0)) {} else { p += (i); goto stop; }
|
||||
NLOHMANN_VIEW_REPEAT16(NLOHMANN_VIEW_STEP)
|
||||
NLOHMANN_VIEW_STEP(0) NLOHMANN_VIEW_STEP(1) NLOHMANN_VIEW_STEP(2) NLOHMANN_VIEW_STEP(3)
|
||||
NLOHMANN_VIEW_STEP(4) NLOHMANN_VIEW_STEP(5) NLOHMANN_VIEW_STEP(6) NLOHMANN_VIEW_STEP(7)
|
||||
if (!Value || !NLOHMANN_VIEW_VECTOR)
|
||||
{
|
||||
NLOHMANN_VIEW_STEP(8) NLOHMANN_VIEW_STEP(9) NLOHMANN_VIEW_STEP(10) NLOHMANN_VIEW_STEP(11)
|
||||
NLOHMANN_VIEW_STEP(12) NLOHMANN_VIEW_STEP(13) NLOHMANN_VIEW_STEP(14) NLOHMANN_VIEW_STEP(15)
|
||||
p += 8;
|
||||
}
|
||||
#undef NLOHMANN_VIEW_STEP
|
||||
p += 16;
|
||||
p += 8;
|
||||
#if NLOHMANN_VIEW_VECTOR
|
||||
p = vector_plain_run(p, e);
|
||||
if (p != e && plain[*p] == 0)
|
||||
{
|
||||
goto stop;
|
||||
}
|
||||
#else
|
||||
while (e - p >= 8)
|
||||
{
|
||||
const std::uint64_t special = swar_string_special(read_eight_bytes(p));
|
||||
@@ -86,6 +104,7 @@ NLOHMANN_VIEW_ALWAYS_INLINE const unsigned char* scan_string_run(const unsigned
|
||||
}
|
||||
p += 8;
|
||||
}
|
||||
#endif
|
||||
continue;
|
||||
}
|
||||
while (p != e && plain[*p] != 0)
|
||||
@@ -101,6 +120,10 @@ stop:
|
||||
{
|
||||
return p; // quote, backslash, or control character
|
||||
}
|
||||
#if NLOHMANN_VIEW_VECTOR_UTF8
|
||||
// non-ASCII: the vector check, out of line
|
||||
return scan_string_vector(p, e, plain);
|
||||
#else
|
||||
// non-ASCII: a run of well-formed sequences (the library's check, so
|
||||
// that exactly what json::parse accepts is accepted)
|
||||
do
|
||||
@@ -113,6 +136,7 @@ stop:
|
||||
p += n;
|
||||
}
|
||||
while (p != e && *p >= 0x80);
|
||||
#endif
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,281 @@
|
||||
// __ _____ _____ _____
|
||||
// __| | __| | | | JSON for Modern C++
|
||||
// | | |__ | | | | | | version 3.12.0
|
||||
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
|
||||
//
|
||||
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
|
||||
// SPDX-FileCopyrightText: 2018-2025 The simdjson authors <https://github.com/simdjson/simdjson>
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <array> // array
|
||||
#include <cstddef> // size_t
|
||||
#include <cstdint> // uint8_t, uint64_t
|
||||
|
||||
#include <nlohmann/json.hpp>
|
||||
#include <nlohmann/detail/view/macro_scope.hpp>
|
||||
|
||||
// Vector code for long runs of string bytes. NEON (AArch64) and SSE2 (x86-64)
|
||||
// belong to the baseline instruction sets and are used by default. The vector
|
||||
// UTF-8 check needs NEON, or SSSE3 if JSON_VIEW_USE_SSSE3 is defined: SSSE3 is
|
||||
// not part of x86-64, so it must not depend on the flags of a translation unit
|
||||
// (two translation units with different flags would have different
|
||||
// definitions of the same inline functions). JSON_VIEW_NO_SIMD selects the
|
||||
// portable code.
|
||||
#if !defined(JSON_VIEW_NO_SIMD) && defined(__aarch64__) && (defined(__GNUC__) || defined(__clang__)) && NLOHMANN_VIEW_LITTLE_ENDIAN
|
||||
#include <arm_neon.h>
|
||||
#define NLOHMANN_VIEW_NEON 1
|
||||
#else
|
||||
#define NLOHMANN_VIEW_NEON 0
|
||||
#endif
|
||||
#if !defined(JSON_VIEW_NO_SIMD) && !NLOHMANN_VIEW_NEON && (defined(__SSE2__) || defined(_M_X64) || (defined(_M_IX86_FP) && _M_IX86_FP >= 2))
|
||||
#include <emmintrin.h>
|
||||
#define NLOHMANN_VIEW_SSE2 1
|
||||
#else
|
||||
#define NLOHMANN_VIEW_SSE2 0
|
||||
#endif
|
||||
#if NLOHMANN_VIEW_SSE2 && defined(JSON_VIEW_USE_SSSE3)
|
||||
#include <tmmintrin.h>
|
||||
#define NLOHMANN_VIEW_SSSE3 1 // NOLINT(cppcoreguidelines-macro-to-enum,modernize-macro-to-enum)
|
||||
#else
|
||||
#define NLOHMANN_VIEW_SSSE3 0 // NOLINT(cppcoreguidelines-macro-to-enum,modernize-macro-to-enum)
|
||||
#endif
|
||||
#define NLOHMANN_VIEW_VECTOR (NLOHMANN_VIEW_NEON || NLOHMANN_VIEW_SSE2)
|
||||
#define NLOHMANN_VIEW_VECTOR_UTF8 (NLOHMANN_VIEW_NEON || NLOHMANN_VIEW_SSSE3)
|
||||
|
||||
NLOHMANN_JSON_NAMESPACE_BEGIN
|
||||
namespace detail
|
||||
{
|
||||
namespace view
|
||||
{
|
||||
|
||||
#if NLOHMANN_VIEW_VECTOR
|
||||
/*!
|
||||
@brief the first byte of a string run that is a quote, a backslash, a control
|
||||
character, or not ASCII, 16 bytes per step
|
||||
|
||||
Stops at such a byte, or where fewer than 16 bytes are left (the caller tells
|
||||
the two apart). A signed compare with 0x20 finds control characters and
|
||||
non-ASCII bytes at once.
|
||||
*/
|
||||
NLOHMANN_VIEW_ALWAYS_INLINE const unsigned char* vector_plain_run(const unsigned char* p, const unsigned char* e) noexcept
|
||||
{
|
||||
while (e - p >= 16)
|
||||
{
|
||||
#if NLOHMANN_VIEW_NEON
|
||||
const uint8x16_t in = vld1q_u8(p);
|
||||
const uint8x16_t special = vorrq_u8(vorrq_u8(vceqq_u8(in, vdupq_n_u8('"')), vceqq_u8(in, vdupq_n_u8('\\'))),
|
||||
vcltq_s8(vreinterpretq_s8_u8(in), vdupq_n_s8(0x20)));
|
||||
// one nibble per byte (the usual NEON replacement of x86's movemask, see
|
||||
// D. Kutenin, "Porting x86 vector bitmask optimizations to Arm NEON", 2022)
|
||||
const std::uint64_t bits = vget_lane_u64(vreinterpret_u64_u8(vshrn_n_u16(vreinterpretq_u16_u8(special), 4)), 0);
|
||||
if (bits != 0)
|
||||
{
|
||||
return p + (count_trailing_zeros(bits) >> 2u);
|
||||
}
|
||||
#else
|
||||
const __m128i in = _mm_loadu_si128(static_cast<const __m128i*>(static_cast<const void*>(p)));
|
||||
const __m128i special = _mm_or_si128(_mm_or_si128(_mm_cmpeq_epi8(in, _mm_set1_epi8('"')), _mm_cmpeq_epi8(in, _mm_set1_epi8('\\'))),
|
||||
_mm_cmplt_epi8(in, _mm_set1_epi8(0x20)));
|
||||
const auto bits = static_cast<std::uint64_t>(static_cast<unsigned>(_mm_movemask_epi8(special)));
|
||||
if (bits != 0)
|
||||
{
|
||||
return p + count_trailing_zeros(bits);
|
||||
}
|
||||
#endif
|
||||
p += 16;
|
||||
}
|
||||
return p;
|
||||
}
|
||||
#endif
|
||||
|
||||
#if NLOHMANN_VIEW_VECTOR_UTF8
|
||||
/// Tables of the UTF-8 check of J. Keiser and D. Lemire, "Validating UTF-8 In
|
||||
/// Less Than One Instruction Per Byte" (2021), as in simdjson ("lookup4"): each
|
||||
/// maps a nibble (high and low nibble of the previous byte, high nibble of the
|
||||
/// current byte) to the errors it allows; a byte pair is ill-formed if all
|
||||
/// three have an error bit in common.
|
||||
template<typename Dummy = void>
|
||||
struct utf8_lookup4
|
||||
{
|
||||
static constexpr std::uint8_t too_short = 1u << 0u, too_long = 1u << 1u, overlong_3 = 1u << 2u, too_large = 1u << 3u;
|
||||
static constexpr std::uint8_t surrogate = 1u << 4u, overlong_2 = 1u << 5u, too_large_1000 = 1u << 6u, overlong_4 = 1u << 6u;
|
||||
static constexpr std::uint8_t two_conts = 1u << 7u, carry = too_short | too_long | two_conts;
|
||||
static const std::array<std::uint8_t, 16> byte_1_high;
|
||||
static const std::array<std::uint8_t, 16> byte_1_low;
|
||||
static const std::array<std::uint8_t, 16> byte_2_high;
|
||||
};
|
||||
|
||||
template<typename Dummy>
|
||||
const std::array<std::uint8_t, 16> utf8_lookup4<Dummy>::byte_1_high =
|
||||
{
|
||||
{
|
||||
too_long, too_long, too_long, too_long, too_long, too_long, too_long, too_long,
|
||||
two_conts, two_conts, two_conts, two_conts,
|
||||
too_short | overlong_2, too_short, too_short | overlong_3 | surrogate, too_short | too_large | too_large_1000 | overlong_4
|
||||
}
|
||||
};
|
||||
|
||||
template<typename Dummy>
|
||||
const std::array<std::uint8_t, 16> utf8_lookup4<Dummy>::byte_1_low =
|
||||
{
|
||||
{
|
||||
carry | overlong_3 | overlong_2 | overlong_4, carry | overlong_2, carry, carry,
|
||||
carry | too_large, carry | too_large | too_large_1000, carry | too_large | too_large_1000, carry | too_large | too_large_1000,
|
||||
carry | too_large | too_large_1000, carry | too_large | too_large_1000, carry | too_large | too_large_1000, carry | too_large | too_large_1000,
|
||||
carry | too_large | too_large_1000, carry | too_large | too_large_1000 | surrogate, carry | too_large | too_large_1000, carry | too_large | too_large_1000
|
||||
}
|
||||
};
|
||||
|
||||
template<typename Dummy>
|
||||
const std::array<std::uint8_t, 16> utf8_lookup4<Dummy>::byte_2_high =
|
||||
{
|
||||
{
|
||||
too_short, too_short, too_short, too_short, too_short, too_short, too_short, too_short,
|
||||
static_cast<std::uint8_t>(too_long | overlong_2 | two_conts | overlong_3 | too_large_1000 | overlong_4),
|
||||
static_cast<std::uint8_t>(too_long | overlong_2 | two_conts | overlong_3 | too_large),
|
||||
static_cast<std::uint8_t>(too_long | overlong_2 | two_conts | surrogate | too_large),
|
||||
static_cast<std::uint8_t>(too_long | overlong_2 | two_conts | surrogate | too_large),
|
||||
too_short, too_short, too_short, too_short
|
||||
}
|
||||
};
|
||||
|
||||
/// the end of scan_string_vector from block, where the vector loop stopped
|
||||
/// (ill-formed UTF-8, or fewer than 16 bytes left): one byte or sequence at a
|
||||
/// time, from the start of a sequence that crosses into the block
|
||||
inline const unsigned char* scan_string_finish(const unsigned char* p, const unsigned char* block, const unsigned char* e, const std::uint8_t* plain) noexcept
|
||||
{
|
||||
for (int i = 1; i <= 3 && block - i >= p; ++i)
|
||||
{
|
||||
const unsigned char c = block[-i];
|
||||
if (c < 0x80)
|
||||
{
|
||||
break;
|
||||
}
|
||||
if (c >= 0xC0)
|
||||
{
|
||||
const int len = 2 + static_cast<int>(c >= 0xE0) + static_cast<int>(c >= 0xF0);
|
||||
if (len > i)
|
||||
{
|
||||
block -= i;
|
||||
}
|
||||
break;
|
||||
}
|
||||
}
|
||||
for (p = block; p != e;)
|
||||
{
|
||||
if (*p < 0x80)
|
||||
{
|
||||
if (plain[*p] == 0)
|
||||
{
|
||||
return p;
|
||||
}
|
||||
++p;
|
||||
continue;
|
||||
}
|
||||
const std::size_t n = validate_one_utf8(p, static_cast<std::size_t>(e - p));
|
||||
if (n == 0)
|
||||
{
|
||||
return p;
|
||||
}
|
||||
p += n;
|
||||
}
|
||||
return p;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief the rest of a string from p (a character boundary), 16 bytes per step
|
||||
|
||||
The first quote, backslash, or control character is found with vector
|
||||
compares, and the UTF-8 check covers the bytes up to it. Returns where the
|
||||
string scan stops, like scan_string_run: before ill-formed UTF-8 and for the
|
||||
last bytes of the input, the bytes are checked one sequence at a time. Out of
|
||||
line, so that no constants of the check occupy registers in the parse loop.
|
||||
*/
|
||||
NLOHMANN_VIEW_NOINLINE inline const unsigned char* scan_string_vector(const unsigned char* p, const unsigned char* e, const std::uint8_t* plain) noexcept
|
||||
{
|
||||
using lookup = utf8_lookup4<>;
|
||||
const unsigned char* block = p;
|
||||
#if NLOHMANN_VIEW_NEON
|
||||
const uint8x16_t t1h = vld1q_u8(lookup::byte_1_high.data());
|
||||
const uint8x16_t t1l = vld1q_u8(lookup::byte_1_low.data());
|
||||
const uint8x16_t t2h = vld1q_u8(lookup::byte_2_high.data());
|
||||
uint8x16_t prev = vdupq_n_u8(0);
|
||||
while (e - block >= 16)
|
||||
{
|
||||
const uint8x16_t in = vld1q_u8(block);
|
||||
const uint8x16_t special = vorrq_u8(vorrq_u8(vceqq_u8(in, vdupq_n_u8('"')), vceqq_u8(in, vdupq_n_u8('\\'))), vcltq_u8(in, vdupq_n_u8(0x20)));
|
||||
const uint8x16_t prev1 = vextq_u8(prev, in, 15);
|
||||
const uint8x16_t sc = vandq_u8(vandq_u8(vqtbl1q_u8(t1h, vshrq_n_u8(prev1, 4)), vqtbl1q_u8(t1l, vandq_u8(prev1, vdupq_n_u8(0x0F)))), vqtbl1q_u8(t2h, vshrq_n_u8(in, 4)));
|
||||
const uint8x16_t must23 = vorrq_u8(vqsubq_u8(vextq_u8(prev, in, 14), vdupq_n_u8(0xE0 - 0x80)), vqsubq_u8(vextq_u8(prev, in, 13), vdupq_n_u8(0xF0 - 0x80)));
|
||||
const uint8x16_t err = veorq_u8(vandq_u8(must23, vdupq_n_u8(0x80)), sc);
|
||||
const std::uint64_t special_bits = vget_lane_u64(vreinterpret_u64_u8(vshrn_n_u16(vreinterpretq_u16_u8(special), 4)), 0);
|
||||
const std::uint64_t err_bits = vget_lane_u64(vreinterpret_u64_u8(vshrn_n_u16(vreinterpretq_u16_u8(vtstq_u8(err, err)), 4)), 0);
|
||||
if (special_bits != 0)
|
||||
{
|
||||
// errors up to the special byte count (an incomplete sequence
|
||||
// before a quote shows at the quote); the bytes after it do not
|
||||
const unsigned k = static_cast<unsigned>(count_trailing_zeros(special_bits)) >> 2u;
|
||||
const std::uint64_t upto = k == 15 ? ~std::uint64_t{0} :
|
||||
(std::uint64_t{1} << (4u * (k + 1u))) - 1u;
|
||||
if ((err_bits & upto) == 0)
|
||||
{
|
||||
return block + k;
|
||||
}
|
||||
break;
|
||||
}
|
||||
if (err_bits != 0)
|
||||
{
|
||||
break;
|
||||
}
|
||||
prev = in;
|
||||
block += 16;
|
||||
}
|
||||
#else
|
||||
// the same with SSSE3 (pshufb for the table lookups; nibbles from 16-bit
|
||||
// shifts, as there are no byte shifts)
|
||||
const __m128i t1h = _mm_loadu_si128(static_cast<const __m128i*>(static_cast<const void*>(lookup::byte_1_high.data())));
|
||||
const __m128i t1l = _mm_loadu_si128(static_cast<const __m128i*>(static_cast<const void*>(lookup::byte_1_low.data())));
|
||||
const __m128i t2h = _mm_loadu_si128(static_cast<const __m128i*>(static_cast<const void*>(lookup::byte_2_high.data())));
|
||||
const __m128i nibble = _mm_set1_epi8(0x0F);
|
||||
const __m128i zero = _mm_setzero_si128();
|
||||
__m128i prev = zero;
|
||||
while (e - block >= 16)
|
||||
{
|
||||
const __m128i in = _mm_loadu_si128(static_cast<const __m128i*>(static_cast<const void*>(block)));
|
||||
const __m128i special = _mm_or_si128(_mm_or_si128(_mm_cmpeq_epi8(in, _mm_set1_epi8('"')), _mm_cmpeq_epi8(in, _mm_set1_epi8('\\'))),
|
||||
_mm_cmpeq_epi8(_mm_subs_epu8(in, _mm_set1_epi8(0x1F)), zero)); // in < 0x20
|
||||
const __m128i prev1 = _mm_alignr_epi8(in, prev, 15);
|
||||
const __m128i sc = _mm_and_si128(_mm_and_si128(_mm_shuffle_epi8(t1h, _mm_and_si128(_mm_srli_epi16(prev1, 4), nibble)),
|
||||
_mm_shuffle_epi8(t1l, _mm_and_si128(prev1, nibble))),
|
||||
_mm_shuffle_epi8(t2h, _mm_and_si128(_mm_srli_epi16(in, 4), nibble)));
|
||||
const __m128i must23 = _mm_or_si128(_mm_subs_epu8(_mm_alignr_epi8(in, prev, 14), _mm_set1_epi8(0xE0 - 0x80)),
|
||||
_mm_subs_epu8(_mm_alignr_epi8(in, prev, 13), _mm_set1_epi8(0xF0 - 0x80)));
|
||||
const __m128i err = _mm_xor_si128(_mm_and_si128(must23, _mm_set1_epi8(static_cast<char>(-128))), sc);
|
||||
const auto special_bits = static_cast<unsigned>(_mm_movemask_epi8(special));
|
||||
const auto err_bits = ~static_cast<unsigned>(_mm_movemask_epi8(_mm_cmpeq_epi8(err, zero))) & 0xFFFFu;
|
||||
if (special_bits != 0)
|
||||
{
|
||||
const unsigned k = static_cast<unsigned>(count_trailing_zeros(static_cast<std::uint64_t>(special_bits)));
|
||||
if ((err_bits & ((2u << k) - 1u)) == 0)
|
||||
{
|
||||
return block + k;
|
||||
}
|
||||
break;
|
||||
}
|
||||
if (err_bits != 0)
|
||||
{
|
||||
break;
|
||||
}
|
||||
prev = in;
|
||||
block += 16;
|
||||
}
|
||||
#endif
|
||||
return scan_string_finish(p, block, e, plain);
|
||||
}
|
||||
#endif
|
||||
|
||||
} // namespace view
|
||||
} // namespace detail
|
||||
NLOHMANN_JSON_NAMESPACE_END
|
||||
@@ -25,6 +25,7 @@
|
||||
#define INCLUDE_NLOHMANN_JSON_VIEW_HPP_
|
||||
|
||||
#include <cstddef> // size_t
|
||||
#include <cstdint> // uint32_t
|
||||
#include <cstring> // memcpy, strlen
|
||||
#include <iterator> // distance, input_iterator_tag, iterator_traits
|
||||
#include <map> // map
|
||||
@@ -56,6 +57,7 @@
|
||||
#include <nlohmann/detail/view/macro_scope.hpp>
|
||||
#include <nlohmann/detail/view/materialize.hpp>
|
||||
#include <nlohmann/detail/view/node.hpp>
|
||||
#include <nlohmann/detail/view/object_index.hpp>
|
||||
#include <nlohmann/detail/view/pointer.hpp>
|
||||
#include <nlohmann/detail/view/serializer.hpp>
|
||||
#include <nlohmann/detail/view/string_ref.hpp>
|
||||
@@ -925,7 +927,9 @@ class basic_json_document
|
||||
}
|
||||
return sizeof(document_data) + (m_data->inline_cap * sizeof(detail::view::node))
|
||||
+ (m_data->tape != m_data->inline_tape ? m_data->tape_cap * sizeof(detail::view::node) : 0)
|
||||
+ m_data->arena.capacity() + m_data->owned.capacity();
|
||||
+ m_data->arena.capacity() + m_data->owned.capacity()
|
||||
+ (m_data->indexes.capacity() * sizeof(document_data::object_index)) + (m_data->index_slots.capacity() * sizeof(std::uint32_t))
|
||||
+ (m_data->large_objects.capacity() * sizeof(std::uint32_t));
|
||||
}
|
||||
|
||||
/// release unused capacity of the index and the decoded strings; like
|
||||
@@ -995,6 +999,9 @@ class basic_json_document
|
||||
d.size = size;
|
||||
d.tape_size = 0;
|
||||
d.arena.clear();
|
||||
d.indexes.clear();
|
||||
d.index_slots.clear();
|
||||
d.large_objects.clear();
|
||||
d.discarded = true;
|
||||
detail::view::parse_failure failure;
|
||||
bool ok = false;
|
||||
@@ -1010,6 +1017,7 @@ class basic_json_document
|
||||
{
|
||||
d.base[0] = d.src;
|
||||
d.base[1] = d.arena.data();
|
||||
detail::view::build_object_indexes(d);
|
||||
d.discarded = false;
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -25,6 +25,7 @@
|
||||
#define INCLUDE_NLOHMANN_JSON_VIEW_HPP_
|
||||
|
||||
#include <cstddef> // size_t
|
||||
#include <cstdint> // uint32_t
|
||||
#include <cstring> // memcpy, strlen
|
||||
#include <iterator> // distance, input_iterator_tag, iterator_traits
|
||||
#include <map> // map
|
||||
@@ -81,9 +82,11 @@
|
||||
|
||||
#include <array> // array
|
||||
#include <cstddef> // size_t
|
||||
#include <cstdint> // uint32_t
|
||||
#include <cstring> // memcpy
|
||||
#include <new> // operator new, placement new
|
||||
#include <string> // string
|
||||
#include <vector> // vector
|
||||
|
||||
// #include <nlohmann/json.hpp>
|
||||
// #include <nlohmann/detail/view/macro_scope.hpp>
|
||||
@@ -279,6 +282,17 @@ struct document_data
|
||||
std::size_t inline_cap = 0;
|
||||
std::string arena{}; ///< decoded strings that contained escapes // NOLINT(readability-redundant-member-init)
|
||||
std::string owned{}; ///< owned copy of the input, if any // NOLINT(readability-redundant-member-init)
|
||||
|
||||
// hash indexes of large objects (see object_index.hpp)
|
||||
static constexpr std::uint32_t index_min_members = 128;
|
||||
struct object_index
|
||||
{
|
||||
std::size_t start; ///< first slot in index_slots
|
||||
std::uint32_t mask; ///< slot count - 1 (a power of two minus one)
|
||||
};
|
||||
std::vector<object_index> indexes{}; // NOLINT(readability-redundant-member-init)
|
||||
std::vector<std::uint32_t> index_slots{}; // NOLINT(readability-redundant-member-init)
|
||||
std::vector<std::uint32_t> large_objects{}; ///< positions of the objects to index (noted while parsing) // NOLINT(readability-redundant-member-init)
|
||||
std::array<const char*, 4> base = {{nullptr, nullptr, nullptr, nullptr}}; ///< string bases: source, arena (indexed by flags & node_flags::storage)
|
||||
bool discarded = true;
|
||||
|
||||
@@ -306,7 +320,7 @@ struct document_data
|
||||
}
|
||||
};
|
||||
|
||||
document_data() noexcept = default;
|
||||
document_data() = default;
|
||||
document_data(const document_data&) = delete;
|
||||
document_data(document_data&&) = delete;
|
||||
document_data& operator=(const document_data&) = delete;
|
||||
@@ -395,6 +409,290 @@ NLOHMANN_JSON_NAMESPACE_END
|
||||
// #include <nlohmann/json.hpp>
|
||||
// #include <nlohmann/detail/view/macro_scope.hpp>
|
||||
|
||||
// #include <nlohmann/detail/view/simd.hpp>
|
||||
// __ _____ _____ _____
|
||||
// __| | __| | | | JSON for Modern C++
|
||||
// | | |__ | | | | | | version 3.12.0
|
||||
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
|
||||
//
|
||||
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
|
||||
// SPDX-FileCopyrightText: 2018-2025 The simdjson authors <https://github.com/simdjson/simdjson>
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
|
||||
|
||||
#include <array> // array
|
||||
#include <cstddef> // size_t
|
||||
#include <cstdint> // uint8_t, uint64_t
|
||||
|
||||
// #include <nlohmann/json.hpp>
|
||||
// #include <nlohmann/detail/view/macro_scope.hpp>
|
||||
|
||||
|
||||
// Vector code for long runs of string bytes. NEON (AArch64) and SSE2 (x86-64)
|
||||
// belong to the baseline instruction sets and are used by default. The vector
|
||||
// UTF-8 check needs NEON, or SSSE3 if JSON_VIEW_USE_SSSE3 is defined: SSSE3 is
|
||||
// not part of x86-64, so it must not depend on the flags of a translation unit
|
||||
// (two translation units with different flags would have different
|
||||
// definitions of the same inline functions). JSON_VIEW_NO_SIMD selects the
|
||||
// portable code.
|
||||
#if !defined(JSON_VIEW_NO_SIMD) && defined(__aarch64__) && (defined(__GNUC__) || defined(__clang__)) && NLOHMANN_VIEW_LITTLE_ENDIAN
|
||||
#include <arm_neon.h>
|
||||
#define NLOHMANN_VIEW_NEON 1
|
||||
#else
|
||||
#define NLOHMANN_VIEW_NEON 0
|
||||
#endif
|
||||
#if !defined(JSON_VIEW_NO_SIMD) && !NLOHMANN_VIEW_NEON && (defined(__SSE2__) || defined(_M_X64) || (defined(_M_IX86_FP) && _M_IX86_FP >= 2))
|
||||
#include <emmintrin.h>
|
||||
#define NLOHMANN_VIEW_SSE2 1
|
||||
#else
|
||||
#define NLOHMANN_VIEW_SSE2 0
|
||||
#endif
|
||||
#if NLOHMANN_VIEW_SSE2 && defined(JSON_VIEW_USE_SSSE3)
|
||||
#include <tmmintrin.h>
|
||||
#define NLOHMANN_VIEW_SSSE3 1 // NOLINT(cppcoreguidelines-macro-to-enum,modernize-macro-to-enum)
|
||||
#else
|
||||
#define NLOHMANN_VIEW_SSSE3 0 // NOLINT(cppcoreguidelines-macro-to-enum,modernize-macro-to-enum)
|
||||
#endif
|
||||
#define NLOHMANN_VIEW_VECTOR (NLOHMANN_VIEW_NEON || NLOHMANN_VIEW_SSE2)
|
||||
#define NLOHMANN_VIEW_VECTOR_UTF8 (NLOHMANN_VIEW_NEON || NLOHMANN_VIEW_SSSE3)
|
||||
|
||||
NLOHMANN_JSON_NAMESPACE_BEGIN
|
||||
namespace detail
|
||||
{
|
||||
namespace view
|
||||
{
|
||||
|
||||
#if NLOHMANN_VIEW_VECTOR
|
||||
/*!
|
||||
@brief the first byte of a string run that is a quote, a backslash, a control
|
||||
character, or not ASCII, 16 bytes per step
|
||||
|
||||
Stops at such a byte, or where fewer than 16 bytes are left (the caller tells
|
||||
the two apart). A signed compare with 0x20 finds control characters and
|
||||
non-ASCII bytes at once.
|
||||
*/
|
||||
NLOHMANN_VIEW_ALWAYS_INLINE const unsigned char* vector_plain_run(const unsigned char* p, const unsigned char* e) noexcept
|
||||
{
|
||||
while (e - p >= 16)
|
||||
{
|
||||
#if NLOHMANN_VIEW_NEON
|
||||
const uint8x16_t in = vld1q_u8(p);
|
||||
const uint8x16_t special = vorrq_u8(vorrq_u8(vceqq_u8(in, vdupq_n_u8('"')), vceqq_u8(in, vdupq_n_u8('\\'))),
|
||||
vcltq_s8(vreinterpretq_s8_u8(in), vdupq_n_s8(0x20)));
|
||||
// one nibble per byte (the usual NEON replacement of x86's movemask, see
|
||||
// D. Kutenin, "Porting x86 vector bitmask optimizations to Arm NEON", 2022)
|
||||
const std::uint64_t bits = vget_lane_u64(vreinterpret_u64_u8(vshrn_n_u16(vreinterpretq_u16_u8(special), 4)), 0);
|
||||
if (bits != 0)
|
||||
{
|
||||
return p + (count_trailing_zeros(bits) >> 2u);
|
||||
}
|
||||
#else
|
||||
const __m128i in = _mm_loadu_si128(static_cast<const __m128i*>(static_cast<const void*>(p)));
|
||||
const __m128i special = _mm_or_si128(_mm_or_si128(_mm_cmpeq_epi8(in, _mm_set1_epi8('"')), _mm_cmpeq_epi8(in, _mm_set1_epi8('\\'))),
|
||||
_mm_cmplt_epi8(in, _mm_set1_epi8(0x20)));
|
||||
const auto bits = static_cast<std::uint64_t>(static_cast<unsigned>(_mm_movemask_epi8(special)));
|
||||
if (bits != 0)
|
||||
{
|
||||
return p + count_trailing_zeros(bits);
|
||||
}
|
||||
#endif
|
||||
p += 16;
|
||||
}
|
||||
return p;
|
||||
}
|
||||
#endif
|
||||
|
||||
#if NLOHMANN_VIEW_VECTOR_UTF8
|
||||
/// Tables of the UTF-8 check of J. Keiser and D. Lemire, "Validating UTF-8 In
|
||||
/// Less Than One Instruction Per Byte" (2021), as in simdjson ("lookup4"): each
|
||||
/// maps a nibble (high and low nibble of the previous byte, high nibble of the
|
||||
/// current byte) to the errors it allows; a byte pair is ill-formed if all
|
||||
/// three have an error bit in common.
|
||||
template<typename Dummy = void>
|
||||
struct utf8_lookup4
|
||||
{
|
||||
static constexpr std::uint8_t too_short = 1u << 0u, too_long = 1u << 1u, overlong_3 = 1u << 2u, too_large = 1u << 3u;
|
||||
static constexpr std::uint8_t surrogate = 1u << 4u, overlong_2 = 1u << 5u, too_large_1000 = 1u << 6u, overlong_4 = 1u << 6u;
|
||||
static constexpr std::uint8_t two_conts = 1u << 7u, carry = too_short | too_long | two_conts;
|
||||
static const std::array<std::uint8_t, 16> byte_1_high;
|
||||
static const std::array<std::uint8_t, 16> byte_1_low;
|
||||
static const std::array<std::uint8_t, 16> byte_2_high;
|
||||
};
|
||||
|
||||
template<typename Dummy>
|
||||
const std::array<std::uint8_t, 16> utf8_lookup4<Dummy>::byte_1_high =
|
||||
{
|
||||
{
|
||||
too_long, too_long, too_long, too_long, too_long, too_long, too_long, too_long,
|
||||
two_conts, two_conts, two_conts, two_conts,
|
||||
too_short | overlong_2, too_short, too_short | overlong_3 | surrogate, too_short | too_large | too_large_1000 | overlong_4
|
||||
}
|
||||
};
|
||||
|
||||
template<typename Dummy>
|
||||
const std::array<std::uint8_t, 16> utf8_lookup4<Dummy>::byte_1_low =
|
||||
{
|
||||
{
|
||||
carry | overlong_3 | overlong_2 | overlong_4, carry | overlong_2, carry, carry,
|
||||
carry | too_large, carry | too_large | too_large_1000, carry | too_large | too_large_1000, carry | too_large | too_large_1000,
|
||||
carry | too_large | too_large_1000, carry | too_large | too_large_1000, carry | too_large | too_large_1000, carry | too_large | too_large_1000,
|
||||
carry | too_large | too_large_1000, carry | too_large | too_large_1000 | surrogate, carry | too_large | too_large_1000, carry | too_large | too_large_1000
|
||||
}
|
||||
};
|
||||
|
||||
template<typename Dummy>
|
||||
const std::array<std::uint8_t, 16> utf8_lookup4<Dummy>::byte_2_high =
|
||||
{
|
||||
{
|
||||
too_short, too_short, too_short, too_short, too_short, too_short, too_short, too_short,
|
||||
static_cast<std::uint8_t>(too_long | overlong_2 | two_conts | overlong_3 | too_large_1000 | overlong_4),
|
||||
static_cast<std::uint8_t>(too_long | overlong_2 | two_conts | overlong_3 | too_large),
|
||||
static_cast<std::uint8_t>(too_long | overlong_2 | two_conts | surrogate | too_large),
|
||||
static_cast<std::uint8_t>(too_long | overlong_2 | two_conts | surrogate | too_large),
|
||||
too_short, too_short, too_short, too_short
|
||||
}
|
||||
};
|
||||
|
||||
/// the end of scan_string_vector from block, where the vector loop stopped
|
||||
/// (ill-formed UTF-8, or fewer than 16 bytes left): one byte or sequence at a
|
||||
/// time, from the start of a sequence that crosses into the block
|
||||
inline const unsigned char* scan_string_finish(const unsigned char* p, const unsigned char* block, const unsigned char* e, const std::uint8_t* plain) noexcept
|
||||
{
|
||||
for (int i = 1; i <= 3 && block - i >= p; ++i)
|
||||
{
|
||||
const unsigned char c = block[-i];
|
||||
if (c < 0x80)
|
||||
{
|
||||
break;
|
||||
}
|
||||
if (c >= 0xC0)
|
||||
{
|
||||
const int len = 2 + static_cast<int>(c >= 0xE0) + static_cast<int>(c >= 0xF0);
|
||||
if (len > i)
|
||||
{
|
||||
block -= i;
|
||||
}
|
||||
break;
|
||||
}
|
||||
}
|
||||
for (p = block; p != e;)
|
||||
{
|
||||
if (*p < 0x80)
|
||||
{
|
||||
if (plain[*p] == 0)
|
||||
{
|
||||
return p;
|
||||
}
|
||||
++p;
|
||||
continue;
|
||||
}
|
||||
const std::size_t n = validate_one_utf8(p, static_cast<std::size_t>(e - p));
|
||||
if (n == 0)
|
||||
{
|
||||
return p;
|
||||
}
|
||||
p += n;
|
||||
}
|
||||
return p;
|
||||
}
|
||||
|
||||
/*!
|
||||
@brief the rest of a string from p (a character boundary), 16 bytes per step
|
||||
|
||||
The first quote, backslash, or control character is found with vector
|
||||
compares, and the UTF-8 check covers the bytes up to it. Returns where the
|
||||
string scan stops, like scan_string_run: before ill-formed UTF-8 and for the
|
||||
last bytes of the input, the bytes are checked one sequence at a time. Out of
|
||||
line, so that no constants of the check occupy registers in the parse loop.
|
||||
*/
|
||||
NLOHMANN_VIEW_NOINLINE inline const unsigned char* scan_string_vector(const unsigned char* p, const unsigned char* e, const std::uint8_t* plain) noexcept
|
||||
{
|
||||
using lookup = utf8_lookup4<>;
|
||||
const unsigned char* block = p;
|
||||
#if NLOHMANN_VIEW_NEON
|
||||
const uint8x16_t t1h = vld1q_u8(lookup::byte_1_high.data());
|
||||
const uint8x16_t t1l = vld1q_u8(lookup::byte_1_low.data());
|
||||
const uint8x16_t t2h = vld1q_u8(lookup::byte_2_high.data());
|
||||
uint8x16_t prev = vdupq_n_u8(0);
|
||||
while (e - block >= 16)
|
||||
{
|
||||
const uint8x16_t in = vld1q_u8(block);
|
||||
const uint8x16_t special = vorrq_u8(vorrq_u8(vceqq_u8(in, vdupq_n_u8('"')), vceqq_u8(in, vdupq_n_u8('\\'))), vcltq_u8(in, vdupq_n_u8(0x20)));
|
||||
const uint8x16_t prev1 = vextq_u8(prev, in, 15);
|
||||
const uint8x16_t sc = vandq_u8(vandq_u8(vqtbl1q_u8(t1h, vshrq_n_u8(prev1, 4)), vqtbl1q_u8(t1l, vandq_u8(prev1, vdupq_n_u8(0x0F)))), vqtbl1q_u8(t2h, vshrq_n_u8(in, 4)));
|
||||
const uint8x16_t must23 = vorrq_u8(vqsubq_u8(vextq_u8(prev, in, 14), vdupq_n_u8(0xE0 - 0x80)), vqsubq_u8(vextq_u8(prev, in, 13), vdupq_n_u8(0xF0 - 0x80)));
|
||||
const uint8x16_t err = veorq_u8(vandq_u8(must23, vdupq_n_u8(0x80)), sc);
|
||||
const std::uint64_t special_bits = vget_lane_u64(vreinterpret_u64_u8(vshrn_n_u16(vreinterpretq_u16_u8(special), 4)), 0);
|
||||
const std::uint64_t err_bits = vget_lane_u64(vreinterpret_u64_u8(vshrn_n_u16(vreinterpretq_u16_u8(vtstq_u8(err, err)), 4)), 0);
|
||||
if (special_bits != 0)
|
||||
{
|
||||
// errors up to the special byte count (an incomplete sequence
|
||||
// before a quote shows at the quote); the bytes after it do not
|
||||
const unsigned k = static_cast<unsigned>(count_trailing_zeros(special_bits)) >> 2u;
|
||||
const std::uint64_t upto = k == 15 ? ~std::uint64_t{0} :
|
||||
(std::uint64_t{1} << (4u * (k + 1u))) - 1u;
|
||||
if ((err_bits & upto) == 0)
|
||||
{
|
||||
return block + k;
|
||||
}
|
||||
break;
|
||||
}
|
||||
if (err_bits != 0)
|
||||
{
|
||||
break;
|
||||
}
|
||||
prev = in;
|
||||
block += 16;
|
||||
}
|
||||
#else
|
||||
// the same with SSSE3 (pshufb for the table lookups; nibbles from 16-bit
|
||||
// shifts, as there are no byte shifts)
|
||||
const __m128i t1h = _mm_loadu_si128(static_cast<const __m128i*>(static_cast<const void*>(lookup::byte_1_high.data())));
|
||||
const __m128i t1l = _mm_loadu_si128(static_cast<const __m128i*>(static_cast<const void*>(lookup::byte_1_low.data())));
|
||||
const __m128i t2h = _mm_loadu_si128(static_cast<const __m128i*>(static_cast<const void*>(lookup::byte_2_high.data())));
|
||||
const __m128i nibble = _mm_set1_epi8(0x0F);
|
||||
const __m128i zero = _mm_setzero_si128();
|
||||
__m128i prev = zero;
|
||||
while (e - block >= 16)
|
||||
{
|
||||
const __m128i in = _mm_loadu_si128(static_cast<const __m128i*>(static_cast<const void*>(block)));
|
||||
const __m128i special = _mm_or_si128(_mm_or_si128(_mm_cmpeq_epi8(in, _mm_set1_epi8('"')), _mm_cmpeq_epi8(in, _mm_set1_epi8('\\'))),
|
||||
_mm_cmpeq_epi8(_mm_subs_epu8(in, _mm_set1_epi8(0x1F)), zero)); // in < 0x20
|
||||
const __m128i prev1 = _mm_alignr_epi8(in, prev, 15);
|
||||
const __m128i sc = _mm_and_si128(_mm_and_si128(_mm_shuffle_epi8(t1h, _mm_and_si128(_mm_srli_epi16(prev1, 4), nibble)),
|
||||
_mm_shuffle_epi8(t1l, _mm_and_si128(prev1, nibble))),
|
||||
_mm_shuffle_epi8(t2h, _mm_and_si128(_mm_srli_epi16(in, 4), nibble)));
|
||||
const __m128i must23 = _mm_or_si128(_mm_subs_epu8(_mm_alignr_epi8(in, prev, 14), _mm_set1_epi8(0xE0 - 0x80)),
|
||||
_mm_subs_epu8(_mm_alignr_epi8(in, prev, 13), _mm_set1_epi8(0xF0 - 0x80)));
|
||||
const __m128i err = _mm_xor_si128(_mm_and_si128(must23, _mm_set1_epi8(static_cast<char>(-128))), sc);
|
||||
const auto special_bits = static_cast<unsigned>(_mm_movemask_epi8(special));
|
||||
const auto err_bits = ~static_cast<unsigned>(_mm_movemask_epi8(_mm_cmpeq_epi8(err, zero))) & 0xFFFFu;
|
||||
if (special_bits != 0)
|
||||
{
|
||||
const unsigned k = static_cast<unsigned>(count_trailing_zeros(static_cast<std::uint64_t>(special_bits)));
|
||||
if ((err_bits & ((2u << k) - 1u)) == 0)
|
||||
{
|
||||
return block + k;
|
||||
}
|
||||
break;
|
||||
}
|
||||
if (err_bits != 0)
|
||||
{
|
||||
break;
|
||||
}
|
||||
prev = in;
|
||||
block += 16;
|
||||
}
|
||||
#endif
|
||||
return scan_string_finish(p, block, e, plain);
|
||||
}
|
||||
#endif
|
||||
|
||||
} // namespace view
|
||||
} // namespace detail
|
||||
NLOHMANN_JSON_NAMESPACE_END
|
||||
|
||||
|
||||
// Scanning primitives of the view's parser. The unrolled checks at fixed
|
||||
// offsets follow yyjson (https://github.com/ibireme/yyjson, MIT license): the
|
||||
@@ -441,9 +739,12 @@ NLOHMANN_VIEW_ALWAYS_INLINE std::uint16_t load16(const unsigned char* p) noexcep
|
||||
|
||||
/// Advance over plain string bytes and well-formed UTF-8. Stops at a quote,
|
||||
/// a backslash, a control character, ill-formed UTF-8, or the end. The first
|
||||
/// 16 bytes are checked one by one, so that the position advances by
|
||||
/// constants in predicted branches (most strings are short); longer runs
|
||||
/// continue eight bytes at a time.
|
||||
/// bytes are checked one by one, so that the position advances by constants
|
||||
/// in predicted branches: 16 for keys, whose lengths repeat from record to
|
||||
/// record, and 8 for string values (Value) where a vector loop follows, as
|
||||
/// their lengths vary more. Longer runs continue 16 bytes at a time with NEON
|
||||
/// or SSE2, else eight bytes at a time.
|
||||
template<bool Value = false>
|
||||
NLOHMANN_VIEW_ALWAYS_INLINE const unsigned char* scan_string_run(const unsigned char* p, const unsigned char* e) noexcept
|
||||
{
|
||||
const std::uint8_t* plain = string_plain();
|
||||
@@ -452,9 +753,23 @@ NLOHMANN_VIEW_ALWAYS_INLINE const unsigned char* scan_string_run(const unsigned
|
||||
if (e - p >= 16)
|
||||
{
|
||||
#define NLOHMANN_VIEW_STEP(i) if (NLOHMANN_VIEW_LIKELY(plain[p[i]] != 0)) {} else { p += (i); goto stop; }
|
||||
NLOHMANN_VIEW_REPEAT16(NLOHMANN_VIEW_STEP)
|
||||
NLOHMANN_VIEW_STEP(0) NLOHMANN_VIEW_STEP(1) NLOHMANN_VIEW_STEP(2) NLOHMANN_VIEW_STEP(3)
|
||||
NLOHMANN_VIEW_STEP(4) NLOHMANN_VIEW_STEP(5) NLOHMANN_VIEW_STEP(6) NLOHMANN_VIEW_STEP(7)
|
||||
if (!Value || !NLOHMANN_VIEW_VECTOR)
|
||||
{
|
||||
NLOHMANN_VIEW_STEP(8) NLOHMANN_VIEW_STEP(9) NLOHMANN_VIEW_STEP(10) NLOHMANN_VIEW_STEP(11)
|
||||
NLOHMANN_VIEW_STEP(12) NLOHMANN_VIEW_STEP(13) NLOHMANN_VIEW_STEP(14) NLOHMANN_VIEW_STEP(15)
|
||||
p += 8;
|
||||
}
|
||||
#undef NLOHMANN_VIEW_STEP
|
||||
p += 16;
|
||||
p += 8;
|
||||
#if NLOHMANN_VIEW_VECTOR
|
||||
p = vector_plain_run(p, e);
|
||||
if (p != e && plain[*p] == 0)
|
||||
{
|
||||
goto stop;
|
||||
}
|
||||
#else
|
||||
while (e - p >= 8)
|
||||
{
|
||||
const std::uint64_t special = swar_string_special(read_eight_bytes(p));
|
||||
@@ -465,6 +780,7 @@ NLOHMANN_VIEW_ALWAYS_INLINE const unsigned char* scan_string_run(const unsigned
|
||||
}
|
||||
p += 8;
|
||||
}
|
||||
#endif
|
||||
continue;
|
||||
}
|
||||
while (p != e && plain[*p] != 0)
|
||||
@@ -480,6 +796,10 @@ stop:
|
||||
{
|
||||
return p; // quote, backslash, or control character
|
||||
}
|
||||
#if NLOHMANN_VIEW_VECTOR_UTF8
|
||||
// non-ASCII: the vector check, out of line
|
||||
return scan_string_vector(p, e, plain);
|
||||
#else
|
||||
// non-ASCII: a run of well-formed sequences (the library's check, so
|
||||
// that exactly what json::parse accepts is accepted)
|
||||
do
|
||||
@@ -492,6 +812,7 @@ stop:
|
||||
p += n;
|
||||
}
|
||||
while (p != e && *p >= 0x80);
|
||||
#endif
|
||||
}
|
||||
}
|
||||
|
||||
@@ -660,6 +981,13 @@ class builder
|
||||
frame shallow[64]; // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays): not initialized on purpose; filled as containers open
|
||||
std::vector<frame> deep{};
|
||||
|
||||
/// remember an object to index after parsing (out of line, so that the
|
||||
/// parse loop only has a call for it)
|
||||
NLOHMANN_VIEW_NOINLINE void note_large_object(std::uint32_t idx)
|
||||
{
|
||||
doc.large_objects.push_back(idx);
|
||||
}
|
||||
|
||||
NLOHMANN_VIEW_NOINLINE bool fail(error_code c, const unsigned char* at) noexcept
|
||||
{
|
||||
m_failure.code = c;
|
||||
@@ -1047,7 +1375,7 @@ class builder
|
||||
switch (cur()) \
|
||||
{ \
|
||||
case '"': \
|
||||
if (NLOHMANN_VIEW_UNLIKELY(!string())) { return false; } \
|
||||
if (NLOHMANN_VIEW_UNLIKELY(!string<true>())) { return false; } \
|
||||
goto NEXT; \
|
||||
case '{': \
|
||||
open(value_t::object); \
|
||||
@@ -1131,7 +1459,7 @@ obj_key:
|
||||
{
|
||||
return fail(error_code::expected_key);
|
||||
}
|
||||
if (NLOHMANN_VIEW_UNLIKELY(!string()))
|
||||
if (NLOHMANN_VIEW_UNLIKELY(!string<false>()))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
@@ -1172,19 +1500,27 @@ obj_next:
|
||||
if (enabled(TrailingCommas) && cur() == '}')
|
||||
{
|
||||
++p;
|
||||
goto close_container;
|
||||
goto close_object;
|
||||
}
|
||||
goto obj_key;
|
||||
}
|
||||
if (cur() == '}')
|
||||
{
|
||||
++p;
|
||||
goto close_container;
|
||||
goto close_object;
|
||||
}
|
||||
return fail(error_code::expected_object_end);
|
||||
|
||||
#undef NLOHMANN_VIEW_VALUE
|
||||
|
||||
close_object:
|
||||
// a large object gets a hash index (objects only, so that closing
|
||||
// an array pays nothing for this)
|
||||
if (NLOHMANN_VIEW_UNLIKELY(cur_count >= document_data::index_min_members))
|
||||
{
|
||||
cold.note_large_object(cur_idx);
|
||||
}
|
||||
|
||||
close_container:
|
||||
close();
|
||||
if (NLOHMANN_VIEW_UNLIKELY(depth == 0))
|
||||
@@ -1234,7 +1570,7 @@ root_done:
|
||||
switch (cur())
|
||||
{
|
||||
case '"':
|
||||
return string();
|
||||
return string<true>();
|
||||
case 't':
|
||||
return literal("true", 4, value_t::boolean, node_flags::is_true);
|
||||
case 'f':
|
||||
@@ -1521,12 +1857,13 @@ indent_done:
|
||||
return true;
|
||||
}
|
||||
|
||||
/// a string (value or key) at p
|
||||
/// a string at p: a value (Value) or a key
|
||||
template<bool Value>
|
||||
NLOHMANN_VIEW_ALWAYS_INLINE bool string()
|
||||
{
|
||||
++p; // opening quote
|
||||
const unsigned char* const s = p;
|
||||
p = scan_string_run(p, e);
|
||||
p = scan_string_run<Value>(p, e);
|
||||
if (NLOHMANN_VIEW_LIKELY(p != e && *p == '"'))
|
||||
{
|
||||
emit(value_t::string, 0, 0, static_cast<std::size_t>(s - b), static_cast<std::uint64_t>(p - s));
|
||||
@@ -2379,6 +2716,142 @@ NLOHMANN_JSON_NAMESPACE_END
|
||||
|
||||
// #include <nlohmann/detail/view/node.hpp>
|
||||
|
||||
// #include <nlohmann/detail/view/object_index.hpp>
|
||||
// __ _____ _____ _____
|
||||
// __| | __| | | | JSON for Modern C++
|
||||
// | | |__ | | | | | | version 3.12.0
|
||||
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
|
||||
//
|
||||
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
|
||||
|
||||
#include <cstddef> // size_t
|
||||
#include <cstdint> // uint32_t, uint64_t
|
||||
#include <cstring> // memcmp
|
||||
|
||||
// #include <nlohmann/json.hpp>
|
||||
// #include <nlohmann/detail/view/document_data.hpp>
|
||||
|
||||
// #include <nlohmann/detail/view/macro_scope.hpp>
|
||||
|
||||
// #include <nlohmann/detail/view/node.hpp>
|
||||
|
||||
|
||||
// Hash indexes of large objects, so that a lookup does not compare thousands
|
||||
// of keys (as Boost.JSON switches from a linear search to a hash table for
|
||||
// large objects). An object with document_data::index_min_members members or
|
||||
// more gets an open-addressing table after parsing; its node stores the
|
||||
// number of the table (1-based) in `extra`. A slot holds the offset of a key
|
||||
// node from its object node (0: empty). Of duplicate keys, the first is kept,
|
||||
// as for the linear search.
|
||||
|
||||
NLOHMANN_JSON_NAMESPACE_BEGIN
|
||||
namespace detail
|
||||
{
|
||||
namespace view
|
||||
{
|
||||
|
||||
/// hash of a key: its bytes, eight at a time, in a fixed byte order
|
||||
inline std::uint64_t key_hash(const char* s, std::size_t n) noexcept
|
||||
{
|
||||
std::uint64_t h = 0x9E3779B97F4A7C15u * (n + 1);
|
||||
const auto* p = reinterpret_cast<const unsigned char*>(s); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
|
||||
while (n >= 8)
|
||||
{
|
||||
h = (h ^ read_eight_bytes(p)) * 0xBF58476D1CE4E5B9u;
|
||||
h ^= h >> 29u;
|
||||
p += 8;
|
||||
n -= 8;
|
||||
}
|
||||
std::uint64_t w = 0;
|
||||
for (std::size_t i = 0; i < n; ++i)
|
||||
{
|
||||
w |= static_cast<std::uint64_t>(p[i]) << (8u * i);
|
||||
}
|
||||
h = (h ^ w) * 0x94D049BB133111EBu;
|
||||
return h ^ (h >> 31u);
|
||||
}
|
||||
|
||||
/// build the table of a large object
|
||||
inline void build_object_index(document_data& d, node* obj)
|
||||
{
|
||||
if (d.indexes.size() >= 0xFFFFu)
|
||||
{
|
||||
return; // LCOV_EXCL_LINE (the number must fit `extra`; more large objects are searched linearly)
|
||||
}
|
||||
std::size_t cap = 16;
|
||||
while (cap < 2 * static_cast<std::size_t>(obj->len))
|
||||
{
|
||||
cap *= 2;
|
||||
}
|
||||
const std::size_t start = d.index_slots.size();
|
||||
d.index_slots.resize(start + cap, 0);
|
||||
std::uint32_t* const slots = d.index_slots.data() + start;
|
||||
const std::size_t mask = cap - 1;
|
||||
for (const node* k = document_data::first_child(obj), *end = document_data::child_end(obj); k != end; k = document_data::after(k + 1))
|
||||
{
|
||||
const char* const key = d.str(*k);
|
||||
const std::uint64_t hash = key_hash(key, k->len); // (a cast of the call would be useless where std::uint64_t is std::size_t)
|
||||
std::size_t i = static_cast<std::size_t>(hash) & mask;
|
||||
bool duplicate = false;
|
||||
while (slots[i] != 0)
|
||||
{
|
||||
const node* const other = obj + slots[i];
|
||||
if (other->len == k->len && (k->len == 0 || std::memcmp(d.str(*other), key, k->len) == 0))
|
||||
{
|
||||
duplicate = true; // keep the first
|
||||
break;
|
||||
}
|
||||
i = (i + 1) & mask;
|
||||
}
|
||||
if (!duplicate)
|
||||
{
|
||||
slots[i] = static_cast<std::uint32_t>(k - obj);
|
||||
}
|
||||
}
|
||||
d.indexes.push_back(document_data::object_index{start, static_cast<std::uint32_t>(mask)});
|
||||
obj->extra = static_cast<std::uint16_t>(d.indexes.size());
|
||||
}
|
||||
|
||||
/// build the tables of the large objects the parser noted
|
||||
inline void build_object_indexes(document_data& d)
|
||||
{
|
||||
for (const std::uint32_t i : d.large_objects)
|
||||
{
|
||||
build_object_index(d, d.tape + i);
|
||||
}
|
||||
}
|
||||
|
||||
/// the key node of the first member with this key of an indexed object, or
|
||||
/// nullptr
|
||||
inline const node* find_indexed(const document_data& d, const node* obj, const char* key, std::size_t n) noexcept
|
||||
{
|
||||
const document_data::object_index& ix = d.indexes[obj->extra - 1u];
|
||||
const std::uint32_t* const slots = d.index_slots.data() + ix.start;
|
||||
const std::uint64_t hash = key_hash(key, n); // (a cast of the call would be useless where std::uint64_t is std::size_t)
|
||||
std::size_t i = static_cast<std::size_t>(hash) & ix.mask;
|
||||
for (;;)
|
||||
{
|
||||
const std::uint32_t s = slots[i];
|
||||
if (s == 0)
|
||||
{
|
||||
return nullptr;
|
||||
}
|
||||
const node* const k = obj + s;
|
||||
if (k->len == n && (n == 0 || std::memcmp(d.str(*k), key, n) == 0))
|
||||
{
|
||||
return k;
|
||||
}
|
||||
i = (i + 1) & ix.mask;
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace view
|
||||
} // namespace detail
|
||||
NLOHMANN_JSON_NAMESPACE_END
|
||||
|
||||
|
||||
NLOHMANN_JSON_NAMESPACE_BEGIN
|
||||
namespace detail
|
||||
@@ -2448,6 +2921,10 @@ class short_key
|
||||
/// nullptr; most keys are rejected by their length, from the index alone
|
||||
inline const node* find_member(const document_data& d, const node* object, const char* key, std::size_t n) noexcept
|
||||
{
|
||||
if (NLOHMANN_VIEW_UNLIKELY(object->extra != 0))
|
||||
{
|
||||
return find_indexed(d, object, key, n); // a large object
|
||||
}
|
||||
const node* const end = document_data::child_end(object);
|
||||
const auto* const k = reinterpret_cast<const unsigned char*>(key); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
|
||||
if (NLOHMANN_VIEW_LIKELY(n <= 16))
|
||||
@@ -2807,6 +3284,8 @@ NLOHMANN_JSON_NAMESPACE_END
|
||||
|
||||
// #include <nlohmann/detail/view/node.hpp>
|
||||
|
||||
// #include <nlohmann/detail/view/object_index.hpp>
|
||||
|
||||
// #include <nlohmann/detail/view/pointer.hpp>
|
||||
// __ _____ _____ _____
|
||||
// __| | __| | | | JSON for Modern C++
|
||||
@@ -4501,7 +4980,9 @@ class basic_json_document
|
||||
}
|
||||
return sizeof(document_data) + (m_data->inline_cap * sizeof(detail::view::node))
|
||||
+ (m_data->tape != m_data->inline_tape ? m_data->tape_cap * sizeof(detail::view::node) : 0)
|
||||
+ m_data->arena.capacity() + m_data->owned.capacity();
|
||||
+ m_data->arena.capacity() + m_data->owned.capacity()
|
||||
+ (m_data->indexes.capacity() * sizeof(document_data::object_index)) + (m_data->index_slots.capacity() * sizeof(std::uint32_t))
|
||||
+ (m_data->large_objects.capacity() * sizeof(std::uint32_t));
|
||||
}
|
||||
|
||||
/// release unused capacity of the index and the decoded strings; like
|
||||
@@ -4571,6 +5052,9 @@ class basic_json_document
|
||||
d.size = size;
|
||||
d.tape_size = 0;
|
||||
d.arena.clear();
|
||||
d.indexes.clear();
|
||||
d.index_slots.clear();
|
||||
d.large_objects.clear();
|
||||
d.discarded = true;
|
||||
detail::view::parse_failure failure;
|
||||
bool ok = false;
|
||||
@@ -4586,6 +5070,7 @@ class basic_json_document
|
||||
{
|
||||
d.base[0] = d.src;
|
||||
d.base[1] = d.arena.data();
|
||||
detail::view::build_object_indexes(d);
|
||||
d.discarded = false;
|
||||
return;
|
||||
}
|
||||
@@ -4751,6 +5236,11 @@ class tuple_element<N, ::nlohmann::detail::view::view_item<View>> // NOLINT(cert
|
||||
#undef NLOHMANN_VIEW_THROW
|
||||
#undef NLOHMANN_VIEW_LITTLE_ENDIAN
|
||||
#undef NLOHMANN_VIEW_REPEAT16
|
||||
#undef NLOHMANN_VIEW_NEON
|
||||
#undef NLOHMANN_VIEW_SSE2
|
||||
#undef NLOHMANN_VIEW_SSSE3
|
||||
#undef NLOHMANN_VIEW_VECTOR
|
||||
#undef NLOHMANN_VIEW_VECTOR_UTF8
|
||||
|
||||
|
||||
#endif // INCLUDE_NLOHMANN_JSON_VIEW_HPP_
|
||||
|
||||
@@ -324,6 +324,26 @@ json_test_add_test_for(src/unit-diagnostic-positions.cpp
|
||||
MAIN test_main CXX_STANDARDS ${test_cxx_standards} ${test_force}
|
||||
)
|
||||
|
||||
# the json_view parser again with the portable string scanning instead of
|
||||
# NEON/SSE2, and on x86-64 with the SSSE3 UTF-8 check (JSON_VIEW_USE_SSSE3)
|
||||
json_test_set_test_options(test-json_view_builder_portable
|
||||
COMPILE_DEFINITIONS JSON_VIEW_NO_SIMD
|
||||
)
|
||||
json_test_add_test_for(src/unit-json_view_builder.cpp
|
||||
NAME test-json_view_builder_portable
|
||||
MAIN test_main CXX_STANDARDS ${test_cxx_standards} ${test_force}
|
||||
)
|
||||
if(CMAKE_SYSTEM_PROCESSOR MATCHES "^(x86_64|AMD64|amd64)$" AND NOT MSVC)
|
||||
json_test_set_test_options(test-json_view_builder_ssse3
|
||||
COMPILE_DEFINITIONS JSON_VIEW_USE_SSSE3
|
||||
COMPILE_OPTIONS -mssse3
|
||||
)
|
||||
json_test_add_test_for(src/unit-json_view_builder.cpp
|
||||
NAME test-json_view_builder_ssse3
|
||||
MAIN test_main CXX_STANDARDS ${test_cxx_standards} ${test_force}
|
||||
)
|
||||
endif()
|
||||
|
||||
# *DO NOT* use json_test_set_test_options() below this line
|
||||
|
||||
#############################################################################
|
||||
|
||||
@@ -1277,3 +1277,60 @@ TEST_CASE("json_view comparison")
|
||||
CHECK(a.root() != json_document::parse(other).root());
|
||||
}
|
||||
}
|
||||
|
||||
TEST_CASE("json_view large objects")
|
||||
{
|
||||
// objects with 128 members or more are looked up with a hash index
|
||||
for (const std::size_t members :
|
||||
{
|
||||
127u, 128u, 129u, 10000u
|
||||
})
|
||||
{
|
||||
CAPTURE(members);
|
||||
std::string text = "{";
|
||||
for (std::size_t i = 0; i < members; ++i)
|
||||
{
|
||||
text += (i != 0 ? ",\"" : "\"") + std::string(i % 23, 'k') + std::to_string(i) + (i % 7 == 0 ? "\\n" : "") + "\":" + std::to_string(i);
|
||||
}
|
||||
text += R"(,"":"empty key","k1":"a duplicate of an earlier key"})";
|
||||
const json_document d = json_document::parse(text);
|
||||
const json_view v = d.root();
|
||||
const json j = json::parse(text);
|
||||
for (std::size_t i = 0; i < members; ++i)
|
||||
{
|
||||
const std::string key = std::string(i % 23, 'k') + std::to_string(i) + (i % 7 == 0 ? "\n" : "");
|
||||
CHECK(v[key].get<std::size_t>() == i);
|
||||
CHECK(v.contains(key));
|
||||
CHECK(v.find(key).key() == key);
|
||||
CHECK(v.at(key).get<std::size_t>() == i);
|
||||
CHECK(!v.contains(key + "x"));
|
||||
}
|
||||
CHECK(v[""].get_string() == "empty key");
|
||||
CHECK(v["k1"].get<int>() == 1); // the first of duplicate keys, as for small objects
|
||||
CHECK(!v.contains("missing"));
|
||||
CHECK_THROWS_WITH_AS(v.at("missing"), "[json.exception.out_of_range.403] key 'missing' not found", json::out_of_range&);
|
||||
CHECK(v == j);
|
||||
CHECK(v.materialize() == j);
|
||||
}
|
||||
|
||||
SECTION("nested, reused, and in arrays")
|
||||
{
|
||||
std::string inner = "{";
|
||||
for (int i = 0; i < 300; ++i)
|
||||
{
|
||||
inner += (i != 0 ? ",\"m" : "\"m") + std::to_string(i) + "\":" + std::to_string(i);
|
||||
}
|
||||
inner += '}';
|
||||
const std::string text = "[" + inner + ",{\"x\":" + inner + "}," + inner + "]";
|
||||
json_document d = json_document::parse(text);
|
||||
CHECK(d.root()[0]["m299"].get<int>() == 299);
|
||||
CHECK(d.root()[1]["x"]["m150"].get<int>() == 150);
|
||||
CHECK(d.root()[2]["m0"].get<int>() == 0);
|
||||
const std::size_t with_index = d.memory_usage();
|
||||
d.read(std::string("{\"small\": 1}"));
|
||||
CHECK(d.root()["small"].get<int>() == 1);
|
||||
d.read(text);
|
||||
CHECK(d.root()[2]["m7"].get<int>() == 7);
|
||||
CHECK(d.memory_usage() >= with_index / 2);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -17,6 +17,7 @@
|
||||
#endif
|
||||
using nlohmann::json;
|
||||
|
||||
#include <array>
|
||||
#include <cstdint>
|
||||
#include <fstream>
|
||||
#include <map>
|
||||
@@ -142,6 +143,32 @@ void check_same(const std::string& text)
|
||||
}
|
||||
}
|
||||
|
||||
// a string value and a key must be accepted or rejected as json::parse does,
|
||||
// and give its value (cheaper than check_same: the options do not matter)
|
||||
void check_string(const std::string& content)
|
||||
{
|
||||
for (const std::string& text :
|
||||
{
|
||||
"[\"" + content + "\"]", "{\"" + content + "\":1}"
|
||||
})
|
||||
{
|
||||
const bool accepted = json::accept(text);
|
||||
for (const bool sentinel :
|
||||
{
|
||||
true, false
|
||||
})
|
||||
{
|
||||
const built b = build(text, false, false, sentinel);
|
||||
if (b.ok != accepted || (accepted && value_of(b) != json::parse(text)))
|
||||
{
|
||||
CAPTURE(text);
|
||||
CHECK(b.ok == accepted);
|
||||
CHECK((b.ok && accepted ? value_of(b) == json::parse(text) : true));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// a small deterministic generator of documents
|
||||
struct generator
|
||||
{
|
||||
@@ -401,3 +428,81 @@ TEST_CASE("json_view builder")
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
TEST_CASE("json_view builder: strings across vector blocks")
|
||||
{
|
||||
// Strings are scanned 8 or 16 bytes at a time (NEON, SSE2, or SWAR) and
|
||||
// non-ASCII text with the vector UTF-8 check (NEON, SSSE3) or one sequence
|
||||
// at a time. Sequences are placed so that they start at every offset
|
||||
// around the block boundaries of keys (16, 32) and values (8, 24), with
|
||||
// text of several lengths after them.
|
||||
const std::array<std::size_t, 16> prefixes = {{0, 6, 7, 8, 13, 14, 15, 16, 21, 22, 23, 24, 29, 30, 31, 32}};
|
||||
const std::array<std::size_t, 3> suffixes = {{0, 3, 17}};
|
||||
const auto around = [&](const std::string & seq, std::size_t prefix, std::size_t suffix)
|
||||
{
|
||||
return std::string(prefix, 'a') + seq + std::string(suffix, 'b');
|
||||
};
|
||||
|
||||
SECTION("every two-byte sequence")
|
||||
{
|
||||
for (unsigned lead = 0x80; lead <= 0xFF; ++lead)
|
||||
{
|
||||
for (unsigned second = 0; second <= 0xFF; ++second)
|
||||
{
|
||||
if (second == '"' || second == '\\')
|
||||
{
|
||||
continue;
|
||||
}
|
||||
const std::string seq = {static_cast<char>(lead), static_cast<char>(second)};
|
||||
check_string(around(seq, prefixes[(lead + second) % 16], suffixes[second % 3]));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("three- and four-byte sequences")
|
||||
{
|
||||
const std::array<unsigned, 8> conts = {{0x7F, 0x80, 0x8F, 0x90, 0x9F, 0xA0, 0xBF, 0xC0}};
|
||||
for (unsigned lead = 0xE0; lead <= 0xF7; ++lead)
|
||||
{
|
||||
for (const unsigned b2 : conts)
|
||||
{
|
||||
for (const unsigned b3 : conts)
|
||||
{
|
||||
for (const std::size_t prefix : prefixes)
|
||||
{
|
||||
std::string seq = {static_cast<char>(lead), static_cast<char>(b2), static_cast<char>(b3)};
|
||||
if (lead >= 0xF0)
|
||||
{
|
||||
seq += static_cast<char>(prefix % 2 == 0 ? 0x80 : 0xBF);
|
||||
}
|
||||
check_string(around(seq, prefix, suffixes[prefix % 3]));
|
||||
// cut short before the end of the string
|
||||
check_string(around(seq.substr(0, seq.size() - 1), prefix, suffixes[prefix % 3]));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("long runs of text with one damaged byte")
|
||||
{
|
||||
const std::array<const char*, 7> chars = {{"a", "\xc3\xa9", "\xe3\x81\x82", "\xf0\x9f\x98\x80", "\xed\x9f\xbf", "\xef\xbf\xbf", "\xf4\x8f\xbf\xbf"}};
|
||||
const std::array<char, 11> damage = {{'\x80', '\xbf', '\xc0', '\xc1', '\xe0', '\xed', '\xf5', '\xff', '\x1f', '"', '\\'}};
|
||||
std::mt19937 rng(5295); // NOLINT(cert-msc32-c,cert-msc51-cpp,bugprone-random-generator-seed): reproducible
|
||||
for (int i = 0; i < 4000; ++i)
|
||||
{
|
||||
std::string text;
|
||||
const auto n = rng() % 60;
|
||||
for (unsigned k = 0; k < n; ++k)
|
||||
{
|
||||
text += chars[rng() % chars.size()];
|
||||
}
|
||||
check_string(text);
|
||||
if (!text.empty())
|
||||
{
|
||||
text[rng() % text.size()] = damage[rng() % damage.size()];
|
||||
check_string(text);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user