diff --git a/BUILD.bazel b/BUILD.bazel index 9b76c1c6d..d186ffbbd 100644 --- a/BUILD.bazel +++ b/BUILD.bazel @@ -79,9 +79,11 @@ cc_library( "include/nlohmann/detail/view/materialize.hpp", "include/nlohmann/detail/view/node.hpp", "include/nlohmann/detail/view/number.hpp", + "include/nlohmann/detail/view/object_index.hpp", "include/nlohmann/detail/view/pointer.hpp", "include/nlohmann/detail/view/scan.hpp", "include/nlohmann/detail/view/serializer.hpp", + "include/nlohmann/detail/view/simd.hpp", "include/nlohmann/detail/view/string_ref.hpp", "include/nlohmann/detail/view/value.hpp", "include/nlohmann/json.hpp", diff --git a/README.md b/README.md index e80ede97d..ddf01f6ef 100644 --- a/README.md +++ b/README.md @@ -1404,6 +1404,7 @@ THE SOFTWARE IS PROVIDED “AS IS”, WITHOUT WARRANTY OF ANY KIND, EXPRESS OR I - The class contains parts of [Google Abseil](https://github.com/abseil/abseil-cpp) which is licensed under the [Apache 2.0 License](https://opensource.org/licenses/Apache-2.0). - The class contains an adapted version of the Eisel-Lemire algorithm, its table of powers of five, and its digit comparison for long numbers from [fast_float](https://github.com/fastfloat/fast_float) by Daniel Lemire and contributors, which is available under the [MIT License](https://opensource.org/licenses/MIT) (used here), the Apache 2.0 License, and the Boost Software License. Copyright © 2021 The fast_float authors - The view's parser (``) contains techniques and code adapted from [yyjson](https://github.com/ibireme/yyjson) by YaoYuan, which is licensed under the [MIT License](https://opensource.org/licenses/MIT) (see above): table-driven decoding of `\u` escapes and fixed-offset unrolled checks. +- The view's parser (``) validates non-ASCII strings with the vector UTF-8 check of [simdjson](https://github.com/simdjson/simdjson) by Daniel Lemire, Geoff Langdale, John Keiser, and contributors (its "lookup4" algorithm and tables, after J. Keiser and D. Lemire, "Validating UTF-8 In Less Than One Instruction Per Byte", 2021), which is available under the [MIT License](https://opensource.org/licenses/MIT) (used here) and the Apache 2.0 License. Copyright © 2018-2025 The simdjson authors REUSE Software diff --git a/docs/docset/docSet.sql b/docs/docset/docSet.sql index 2f8685b40..15c24e53d 100644 --- a/docs/docset/docSet.sql +++ b/docs/docset/docSet.sql @@ -348,3 +348,5 @@ INSERT INTO searchIndex(name, type, path) VALUES ('NLOHMANN_JSON_SERIALIZE_ENUM_ INSERT INTO searchIndex(name, type, path) VALUES ('NLOHMANN_JSON_VERSION_MAJOR', 'Macro', 'api/macros/nlohmann_json_version_major/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('NLOHMANN_JSON_VERSION_MINOR', 'Macro', 'api/macros/nlohmann_json_version_major/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('NLOHMANN_JSON_VERSION_PATCH', 'Macro', 'api/macros/nlohmann_json_version_major/index.html'); +INSERT INTO searchIndex(name, type, path) VALUES ('JSON_VIEW_NO_SIMD', 'Macro', 'api/macros/json_view_no_simd/index.html'); +INSERT INTO searchIndex(name, type, path) VALUES ('JSON_VIEW_USE_SSSE3', 'Macro', 'api/macros/json_view_use_ssse3/index.html'); diff --git a/docs/mkdocs/docs/api/basic_json_document/read.md b/docs/mkdocs/docs/api/basic_json_document/read.md index 2267e46bd..2b8e0fc48 100644 --- a/docs/mkdocs/docs/api/basic_json_document/read.md +++ b/docs/mkdocs/docs/api/basic_json_document/read.md @@ -49,6 +49,11 @@ whether or not the new parse succeeds; take fresh views from [`root()`](root.md) `input` is borrowed or owned by the same rules as [`parse()`](parse.md#notes); a document can borrow on one call and own on the next, since ownership is decided freshly each time. +Reusing a document matters most for large inputs: the operating system provides the memory of a fresh node index one +page at a time, and every page costs a page fault the first time it is written. On x86-64 Linux (4 KiB pages), parsing +a 55 MB document into a reused document took about 40 % less time than parsing it into a fresh one. Programs that parse +many documents of similar size should therefore keep one document and call `read()`. + ## Examples ??? example diff --git a/docs/mkdocs/docs/api/basic_json_view/at.md b/docs/mkdocs/docs/api/basic_json_view/at.md index d1b1732bb..589e42b7b 100644 --- a/docs/mkdocs/docs/api/basic_json_view/at.md +++ b/docs/mkdocs/docs/api/basic_json_view/at.md @@ -74,6 +74,8 @@ None of these exceptions carry a [`JSON_DIAGNOSTICS`](../macros/json_diagnostics 1. Linear in the number of members: as for [`ordered_json`](../ordered_json.md), members are compared one after another, in document order, stopping at the first match. Each comparison first checks the key's length -- already known from the index, without reading the key bytes -- before comparing its content. + Objects with 128 or more members get a hash index while parsing, so that a lookup in them takes constant time + on average. 2. Linear in `idx`: elements are skipped one at a time from the first one, since they are not a fixed size in the index (unlike `BasicJsonType`'s array, which is random-access). 3. Linear in the number of reference tokens of `ptr` and, for each token, in the number of members of the object at diff --git a/docs/mkdocs/docs/api/basic_json_view/contains.md b/docs/mkdocs/docs/api/basic_json_view/contains.md index bbd791887..a48395854 100644 --- a/docs/mkdocs/docs/api/basic_json_view/contains.md +++ b/docs/mkdocs/docs/api/basic_json_view/contains.md @@ -35,6 +35,8 @@ No-throw guarantee: this function never throws exceptions. 1. Linear in the number of members: as for [`ordered_json`](../ordered_json.md), members are compared one after another, in document order, stopping at the first match. Each comparison first checks the key's length -- already known from the index, without reading the key bytes -- before comparing its content. + Objects with 128 or more members get a hash index while parsing, so that a lookup in them takes constant time + on average. 2. Linear in the number of reference tokens of `ptr` and, for each token, in the number of members of the object at that level or the index into the array -- as for [`operator[]`](operator[].md#complexity) and [`at`](at.md#complexity) with a JSON pointer. diff --git a/docs/mkdocs/docs/api/basic_json_view/count.md b/docs/mkdocs/docs/api/basic_json_view/count.md index 6a599477e..9cdfe3dee 100644 --- a/docs/mkdocs/docs/api/basic_json_view/count.md +++ b/docs/mkdocs/docs/api/basic_json_view/count.md @@ -26,6 +26,8 @@ No-throw guarantee: this function never throws exceptions. Linear in the number of members: as for [`ordered_json`](../ordered_json.md), members are compared one after another, in document order, stopping at the first match. Each comparison first checks the key's length -- already known from the index, without reading the key bytes -- before comparing its content. +Objects with 128 or more members get a hash index while parsing, so that a lookup in them takes constant time on +average. ## Notes diff --git a/docs/mkdocs/docs/api/basic_json_view/find.md b/docs/mkdocs/docs/api/basic_json_view/find.md index f7ad92e9f..04ee0476b 100644 --- a/docs/mkdocs/docs/api/basic_json_view/find.md +++ b/docs/mkdocs/docs/api/basic_json_view/find.md @@ -28,6 +28,8 @@ No-throw guarantee: this function never throws exceptions. Linear in the number of members: as for [`ordered_json`](../ordered_json.md), members are compared one after another, in document order, stopping at the first match. Each comparison first checks the key's length -- already known from the index, without reading the key bytes -- before comparing its content. +Objects with 128 or more members get a hash index while parsing, so that a lookup in them takes constant time on +average. ## Notes diff --git a/docs/mkdocs/docs/api/basic_json_view/operator[].md b/docs/mkdocs/docs/api/basic_json_view/operator[].md index 2361bf77c..835800a0a 100644 --- a/docs/mkdocs/docs/api/basic_json_view/operator[].md +++ b/docs/mkdocs/docs/api/basic_json_view/operator[].md @@ -76,6 +76,8 @@ None of these exceptions carry a [`JSON_DIAGNOSTICS`](../macros/json_diagnostics another, in document order, stopping at the first match. Each comparison first checks the key's length -- already known from the index, without reading the key bytes -- before comparing its content, so a key of a different length than `key` is rejected without touching the source text. + Objects with 128 or more members get a hash index while parsing, so that a lookup in them takes constant time + on average. 2. Linear in `idx`: elements are skipped one at a time from the first one, since they are not a fixed size in the index (unlike `BasicJsonType`'s array, which is random-access). 3. Linear in the number of reference tokens of `ptr` and, for each token, in the number of members of the object at diff --git a/docs/mkdocs/docs/api/basic_json_view/value.md b/docs/mkdocs/docs/api/basic_json_view/value.md index ff63e32f5..06d5372e5 100644 --- a/docs/mkdocs/docs/api/basic_json_view/value.md +++ b/docs/mkdocs/docs/api/basic_json_view/value.md @@ -70,6 +70,8 @@ None of these exceptions carry a [`JSON_DIAGNOSTICS`](../macros/json_diagnostics 1. Linear in the number of members: as for [`operator[]`](operator[].md#complexity), members are compared one after another, in document order, stopping at the first match. Plus the complexity of converting the found member to `T` (see [`get`](get.md)). + Objects with 128 or more members get a hash index while parsing, so that a lookup in them takes constant time + on average. 2. Linear in the number of reference tokens of `ptr` and, for each token, in the number of members of the object at that level or the index into the array -- as for the [`operator[]`](operator[].md#complexity) and [`at`](at.md#complexity) overloads that take a JSON pointer. Plus the complexity of converting the resolved value diff --git a/docs/mkdocs/docs/api/macros/index.md b/docs/mkdocs/docs/api/macros/index.md index eb2807b69..d91000357 100644 --- a/docs/mkdocs/docs/api/macros/index.md +++ b/docs/mkdocs/docs/api/macros/index.md @@ -37,6 +37,8 @@ header. See also the [macro overview page](../../features/macros.md). - [**JSON_SKIP_UNSUPPORTED_COMPILER_CHECK**](json_skip_unsupported_compiler_check.md) - do not warn about unsupported compilers - [**JSON_USE_GLOBAL_UDLS**](json_use_global_udls.md) - place user-defined string literals (UDLs) into the global namespace - [**JSON_USE_SIMDUTF**](json_use_simdutf.md) - use the simdutf library to accelerate UTF-8 validation +- [**JSON_VIEW_NO_SIMD**](json_view_no_simd.md) - use only portable code in the parser of `json_view.hpp` +- [**JSON_VIEW_USE_SSSE3**](json_view_use_ssse3.md) - validate non-ASCII strings with SSSE3 in the parser of `json_view.hpp` ## Library version diff --git a/docs/mkdocs/docs/api/macros/json_view_no_simd.md b/docs/mkdocs/docs/api/macros/json_view_no_simd.md new file mode 100644 index 000000000..daa982760 --- /dev/null +++ b/docs/mkdocs/docs/api/macros/json_view_no_simd.md @@ -0,0 +1,50 @@ +# JSON_VIEW_NO_SIMD + +```cpp +#define JSON_VIEW_NO_SIMD +``` + +When defined, the parser of [`basic_json_document`](../basic_json_document/index.md) (``) +uses only portable C++ to scan strings. By default, it scans long runs of string bytes 16 at a time with NEON on +AArch64 (with GCC and Clang) and SSE2 on x86-64, which are part of the baseline instruction sets of these +architectures, and validates non-ASCII text with NEON (or SSSE3, see +[`JSON_VIEW_USE_SSSE3`](json_view_use_ssse3.md)). + +The same input is accepted or rejected either way, with the same values, and errors are reported the same way; only +the speed differs. The macro exists for platforms whose compilers lack the intrinsics headers, and to test the portable +code. + +!!! warning "Define consistently" + + The macro selects between two definitions of the same inline functions. It must therefore be defined identically for + **every** translation unit that includes ``; prefer a compile definition on the target. + +## Default definition + +By default, `#!cpp JSON_VIEW_NO_SIMD` is not defined, and the vector code is used where available. + +```cpp +#undef JSON_VIEW_NO_SIMD +``` + +## Examples + +??? example + + The code below uses the portable string scanning of the view. + + ```cpp + #define JSON_VIEW_NO_SIMD + #include + + ... + ``` + +## See also + +- [JSON_VIEW_USE_SSSE3](json_view_use_ssse3.md) - validate non-ASCII strings with SSSE3 on x86-64 +- [json_view](../../features/json_view.md) - the zero-copy view + +## Version history + +- Added in version 3.13.0. diff --git a/docs/mkdocs/docs/api/macros/json_view_use_ssse3.md b/docs/mkdocs/docs/api/macros/json_view_use_ssse3.md new file mode 100644 index 000000000..a240cfd75 --- /dev/null +++ b/docs/mkdocs/docs/api/macros/json_view_use_ssse3.md @@ -0,0 +1,53 @@ +# JSON_VIEW_USE_SSSE3 + +```cpp +#define JSON_VIEW_USE_SSSE3 +``` + +When defined on x86-64, the parser of [`basic_json_document`](../basic_json_document/index.md) +(``) validates non-ASCII text in strings with SSSE3 without asking the CPU first. + +By default, the parser checks once at run time whether the CPU has SSSE3 (all x86-64 CPUs since about 2011 have it) +and then validates non-ASCII text 16 bytes at a time, using the "lookup4" algorithm of +[simdjson](https://github.com/simdjson/simdjson); on CPUs without SSSE3, it validates one UTF-8 sequence at a time. +The vector check is compiled for SSSE3 with a function attribute (GCC 4.9 and later, Clang), so this needs no compiler +option. With MSVC, the check uses `__cpuid`. On AArch64, the vector check uses NEON and is always on. + +Define the macro only together with a compiler option that enables SSSE3 (e.g. `-mssse3`, or `-march=` with a CPU that +has it), and only for programs that run on such CPUs. It saves the check of the CPU, which costs little. The same +input is accepted or rejected either way; only the speed of non-ASCII text differs. + +!!! warning "Define consistently" + + The macro selects between two definitions of the same inline functions. It must therefore be defined identically, + with the same compiler options, for **every** translation unit that includes ``; mixing + translation units that define it with ones that do not is an ODR violation. Prefer a compile definition on the + target. + +## Default definition + +By default, `#!cpp JSON_VIEW_USE_SSSE3` is not defined. + +```cpp +#undef JSON_VIEW_USE_SSSE3 +``` + +## Examples + +??? example + + With CMake, for a program that only runs on CPUs with SSSE3: + + ```cmake + target_compile_definitions(your_target PRIVATE JSON_VIEW_USE_SSSE3) + target_compile_options(your_target PRIVATE -mssse3) + ``` + +## See also + +- [JSON_VIEW_NO_SIMD](json_view_no_simd.md) - use only portable code in the view's parser +- [JSON_USE_SIMDUTF](json_use_simdutf.md) - validate UTF-8 with simdutf in `basic_json`'s parser + +## Version history + +- Added in version 3.13.0. diff --git a/docs/mkdocs/docs/home/architecture.md b/docs/mkdocs/docs/home/architecture.md index 457f0d536..df1f0e06d 100644 --- a/docs/mkdocs/docs/home/architecture.md +++ b/docs/mkdocs/docs/home/architecture.md @@ -196,7 +196,7 @@ packet-beta |-------|---------|------------|-------------------------------------------------------------------------------------------------------------------------------| | 0 | `kind` | `uint8_t` | the type, numbered as [`value_t`](../api/basic_json/value_t.md): 0 null, 1 object, 2 array, 3 string, 4 boolean, 5 signed integer, 6 unsigned integer, 7 float | | 1 | `flags` | `uint8_t` | bits 0-1: where a string's bytes are (0: the source text, 1: the buffer of decoded strings, for strings with escapes); bit 2: the value of a boolean | -| 2-3 | `extra` | `uint16_t` | numbers: the number of integer digits (low byte) and fraction digits (high byte), 255 for more; otherwise 0 | +| 2-3 | `extra` | `uint16_t` | numbers: the number of integer digits (low byte) and fraction digits (high byte), 255 for more; objects: the number of their hash index (1-based), or 0; otherwise 0 | | 4-7 | `off` | `uint32_t` | where the value starts: the first byte after a string's opening quote (or its position in the buffer of decoded strings), the first byte of a number or literal, the bracket of an array or object | | 8-11 | `len` | `uint32_t` | strings: the length after decoding; floats and literals: the length of the token; arrays and objects: the number of elements | | 12-15 | `next` | `uint32_t` | arrays and objects: the number of nodes of the subtree, including the node itself | @@ -211,6 +211,11 @@ packet-beta subtree is `next` nodes further for an array or object, and the next node otherwise (`document_data::after`). Views step from element to element this way and skip whole subtrees in constant time. - **Offsets** are 32 bits wide, so a document is limited to 4 GiB (`out_of_range.416`). +- **Large objects** (128 members or more) get a hash index after parsing + ([`detail/view/object_index.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/view/object_index.hpp)): + an open-addressing table whose slots hold the distance from the object's node to a key's node, so that a lookup does + not compare every key. The object's `extra` holds the number of its table. Only 65,535 tables fit into `extra`; + objects beyond them are searched linearly. For example, `#!json {"a": [1, 2.5]}` becomes five nodes. Each node's elements follow it, and `next` leads from an array or object past its subtree: diff --git a/docs/mkdocs/docs/home/license.md b/docs/mkdocs/docs/home/license.md index 3327ad791..53b54fc82 100644 --- a/docs/mkdocs/docs/home/license.md +++ b/docs/mkdocs/docs/home/license.md @@ -25,3 +25,5 @@ The class contains a copy of [Hedley](https://nemequ.github.io/hedley/) from Eva The class contains an adapted version of the Eisel-Lemire algorithm, its table of powers of five, and its digit comparison for long numbers from [fast_float](https://github.com/fastfloat/fast_float) by Daniel Lemire and contributors, which is available under the [MIT License](https://opensource.org/licenses/MIT) (used here), the Apache 2.0 License, and the Boost Software License. Copyright © 2021 The fast_float authors The view's parser (``) contains techniques and code adapted from [yyjson](https://github.com/ibireme/yyjson) by YaoYuan, which is licensed under the [MIT License](https://opensource.org/licenses/MIT) (see above): table-driven decoding of `\u` escapes and fixed-offset unrolled checks. + +The view's parser (``) validates non-ASCII strings with the vector UTF-8 check of [simdjson](https://github.com/simdjson/simdjson) by Daniel Lemire, Geoff Langdale, John Keiser, and contributors (its "lookup4" algorithm and tables, after J. Keiser and D. Lemire, "Validating UTF-8 In Less Than One Instruction Per Byte", 2021), which is available under the [MIT License](https://opensource.org/licenses/MIT) (used here) and the Apache 2.0 License. Copyright © 2018-2025 The simdjson authors diff --git a/docs/mkdocs/mkdocs.yml b/docs/mkdocs/mkdocs.yml index d5adb9476..0eec15fe3 100644 --- a/docs/mkdocs/mkdocs.yml +++ b/docs/mkdocs/mkdocs.yml @@ -380,6 +380,8 @@ nav: - 'JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON': api/macros/json_use_legacy_discarded_value_comparison.md - 'JSON_USE_OBJECTS_FOR_ENUM_KEYED_MAPS': api/macros/json_use_objects_for_enum_keyed_maps.md - 'JSON_USE_SIMDUTF': api/macros/json_use_simdutf.md + - 'JSON_VIEW_NO_SIMD': api/macros/json_view_no_simd.md + - 'JSON_VIEW_USE_SSSE3': api/macros/json_view_use_ssse3.md - 'NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE, NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE_WITH_DEFAULT, NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE_ONLY_SERIALIZE, NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE, NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_WITH_DEFAULT, NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_ONLY_SERIALIZE': api/macros/nlohmann_define_derived_type.md - 'NLOHMANN_DEFINE_TYPE_INTRUSIVE, NLOHMANN_DEFINE_TYPE_INTRUSIVE_WITH_DEFAULT, NLOHMANN_DEFINE_TYPE_INTRUSIVE_ONLY_SERIALIZE': api/macros/nlohmann_define_type_intrusive.md - 'NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE, NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE_WITH_DEFAULT, NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE_ONLY_SERIALIZE': api/macros/nlohmann_define_type_non_intrusive.md diff --git a/include/nlohmann/detail/view/builder.hpp b/include/nlohmann/detail/view/builder.hpp index 1cd01fef6..6b2ee0616 100644 --- a/include/nlohmann/detail/view/builder.hpp +++ b/include/nlohmann/detail/view/builder.hpp @@ -116,6 +116,13 @@ class builder frame shallow[64]; // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays): not initialized on purpose; filled as containers open std::vector deep{}; + /// remember an object to index after parsing (out of line, so that the + /// parse loop only has a call for it) + NLOHMANN_VIEW_NOINLINE void note_large_object(std::uint32_t idx) + { + doc.large_objects.push_back(idx); + } + NLOHMANN_VIEW_NOINLINE bool fail(error_code c, const unsigned char* at) noexcept { m_failure.code = c; @@ -501,7 +508,7 @@ class builder switch (cur()) \ { \ case '"': \ - if (NLOHMANN_VIEW_UNLIKELY(!string())) { return false; } \ + if (NLOHMANN_VIEW_UNLIKELY(!string())) { return false; } \ goto NEXT; \ case '{': \ open(value_t::object); \ @@ -585,7 +592,7 @@ obj_key: { return fail(error_code::expected_key); } - if (NLOHMANN_VIEW_UNLIKELY(!string())) + if (NLOHMANN_VIEW_UNLIKELY(!string())) { return false; } @@ -626,19 +633,27 @@ obj_next: if (enabled(TrailingCommas) && cur() == '}') { ++p; - goto close_container; + goto close_object; } goto obj_key; } if (cur() == '}') { ++p; - goto close_container; + goto close_object; } return fail(error_code::expected_object_end); #undef NLOHMANN_VIEW_VALUE +close_object: + // a large object gets a hash index (objects only, so that closing + // an array pays nothing for this) + if (NLOHMANN_VIEW_UNLIKELY(cur_count >= document_data::index_min_members)) + { + cold.note_large_object(cur_idx); + } + close_container: close(); if (NLOHMANN_VIEW_UNLIKELY(depth == 0)) @@ -688,7 +703,7 @@ root_done: switch (cur()) { case '"': - return string(); + return string(); case 't': return literal("true", 4, value_t::boolean, node_flags::is_true); case 'f': @@ -811,14 +826,19 @@ indent_done: const auto idx = static_cast(emit(k, 0, 0, static_cast(p - b), 0) - base); if (depth != 0) { - const frame f = {cur_idx, cur_count, cur_is_object}; if (NLOHMANN_VIEW_LIKELY(depth <= 64)) { - cold.shallow[depth - 1] = f; + // field by field: a frame put together on the stack and + // copied would be read back wider than it was written, + // and that load waits until the stores are done + frame& f = cold.shallow[depth - 1]; + f.idx = cur_idx; + f.count = cur_count; + f.is_object = cur_is_object; } else { - cold.deep.push_back(f); + cold.deep.push_back(frame{cur_idx, cur_count, cur_is_object}); } } ++depth; @@ -834,19 +854,21 @@ indent_done: n.next = static_cast(out - base) - cur_idx; if (--depth != 0) { - frame f{}; if (NLOHMANN_VIEW_LIKELY(depth <= 64)) { - f = cold.shallow[depth - 1]; + const frame& f = cold.shallow[depth - 1]; + cur_idx = f.idx; + cur_count = f.count; + cur_is_object = f.is_object; } else { - f = cold.deep.back(); + const frame f = cold.deep.back(); cold.deep.pop_back(); + cur_idx = f.idx; + cur_count = f.count; + cur_is_object = f.is_object; } - cur_idx = f.idx; - cur_count = f.count; - cur_is_object = f.is_object; } } @@ -975,12 +997,13 @@ indent_done: return true; } - /// a string (value or key) at p + /// a string at p: a value (Value) or a key + template NLOHMANN_VIEW_ALWAYS_INLINE bool string() { ++p; // opening quote const unsigned char* const s = p; - p = scan_string_run(p, e); + p = scan_string_run(p, e); if (NLOHMANN_VIEW_LIKELY(p != e && *p == '"')) { emit(value_t::string, 0, 0, static_cast(s - b), static_cast(p - s)); diff --git a/include/nlohmann/detail/view/document_data.hpp b/include/nlohmann/detail/view/document_data.hpp index d74383a85..fd72ee66e 100644 --- a/include/nlohmann/detail/view/document_data.hpp +++ b/include/nlohmann/detail/view/document_data.hpp @@ -10,9 +10,11 @@ #include // array #include // size_t +#include // uint32_t #include // memcpy #include // operator new, placement new #include // string +#include // vector #include #include @@ -37,6 +39,17 @@ struct document_data std::size_t inline_cap = 0; std::string arena{}; ///< decoded strings that contained escapes // NOLINT(readability-redundant-member-init) std::string owned{}; ///< owned copy of the input, if any // NOLINT(readability-redundant-member-init) + + // hash indexes of large objects (see object_index.hpp) + static constexpr std::uint32_t index_min_members = 128; + struct object_index + { + std::size_t start; ///< first slot in index_slots + std::uint32_t mask; ///< slot count - 1 (a power of two minus one) + }; + std::vector indexes{}; // NOLINT(readability-redundant-member-init) + std::vector index_slots{}; // NOLINT(readability-redundant-member-init) + std::vector large_objects{}; ///< positions of the objects to index (noted while parsing) // NOLINT(readability-redundant-member-init) std::array base = {{nullptr, nullptr, nullptr, nullptr}}; ///< string bases: source, arena (indexed by flags & node_flags::storage) bool discarded = true; @@ -64,7 +77,7 @@ struct document_data } }; - document_data() noexcept = default; + document_data() = default; document_data(const document_data&) = delete; document_data(document_data&&) = delete; document_data& operator=(const document_data&) = delete; diff --git a/include/nlohmann/detail/view/lookup.hpp b/include/nlohmann/detail/view/lookup.hpp index 71338be32..5d50c0667 100644 --- a/include/nlohmann/detail/view/lookup.hpp +++ b/include/nlohmann/detail/view/lookup.hpp @@ -16,6 +16,7 @@ #include #include #include +#include NLOHMANN_JSON_NAMESPACE_BEGIN namespace detail @@ -85,6 +86,10 @@ class short_key /// nullptr; most keys are rejected by their length, from the index alone inline const node* find_member(const document_data& d, const node* object, const char* key, std::size_t n) noexcept { + if (NLOHMANN_VIEW_UNLIKELY(object->extra != 0)) + { + return find_indexed(d, object, key, n); // a large object + } const node* const end = document_data::child_end(object); const auto* const k = reinterpret_cast(key); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) if (NLOHMANN_VIEW_LIKELY(n <= 16)) diff --git a/include/nlohmann/detail/view/macro_unscope.hpp b/include/nlohmann/detail/view/macro_unscope.hpp index 11f71ad1b..656f1d9a7 100644 --- a/include/nlohmann/detail/view/macro_unscope.hpp +++ b/include/nlohmann/detail/view/macro_unscope.hpp @@ -19,3 +19,10 @@ #undef NLOHMANN_VIEW_THROW #undef NLOHMANN_VIEW_LITTLE_ENDIAN #undef NLOHMANN_VIEW_REPEAT16 +#undef NLOHMANN_VIEW_NEON +#undef NLOHMANN_VIEW_SSE2 +#undef NLOHMANN_VIEW_SSSE3 +#undef NLOHMANN_VIEW_SSSE3_DISPATCH +#undef NLOHMANN_VIEW_SSSE3_TARGET +#undef NLOHMANN_VIEW_VECTOR +#undef NLOHMANN_VIEW_VECTOR_UTF8 diff --git a/include/nlohmann/detail/view/object_index.hpp b/include/nlohmann/detail/view/object_index.hpp new file mode 100644 index 000000000..27348d827 --- /dev/null +++ b/include/nlohmann/detail/view/object_index.hpp @@ -0,0 +1,131 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +#include // size_t +#include // uint32_t, uint64_t +#include // memcmp + +#include +#include +#include +#include + +// Hash indexes of large objects, so that a lookup does not compare thousands +// of keys (as Boost.JSON switches from a linear search to a hash table for +// large objects). An object with document_data::index_min_members members or +// more gets an open-addressing table after parsing; its node stores the +// number of the table (1-based) in `extra`. A slot holds the offset of a key +// node from its object node (0: empty). Of duplicate keys, the first is kept, +// as for the linear search. + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ +namespace view +{ + +/// hash of a key: its bytes, eight at a time, in a fixed byte order +inline std::uint64_t key_hash(const char* s, std::size_t n) noexcept +{ + std::uint64_t h = 0x9E3779B97F4A7C15u * (n + 1); + const auto* p = reinterpret_cast(s); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + while (n >= 8) + { + h = (h ^ read_eight_bytes(p)) * 0xBF58476D1CE4E5B9u; + h ^= h >> 29u; + p += 8; + n -= 8; + } + std::uint64_t w = 0; + for (std::size_t i = 0; i < n; ++i) + { + w |= static_cast(p[i]) << (8u * i); + } + h = (h ^ w) * 0x94D049BB133111EBu; + return h ^ (h >> 31u); +} + +/// build the table of a large object +inline void build_object_index(document_data& d, node* obj) +{ + if (d.indexes.size() >= 0xFFFFu) + { + return; // LCOV_EXCL_LINE (the number must fit `extra`; more large objects are searched linearly) + } + std::size_t cap = 16; + while (cap < 2 * static_cast(obj->len)) + { + cap *= 2; + } + const std::size_t start = d.index_slots.size(); + d.index_slots.resize(start + cap, 0); + std::uint32_t* const slots = d.index_slots.data() + start; + const std::size_t mask = cap - 1; + for (const node* k = document_data::first_child(obj), *end = document_data::child_end(obj); k != end; k = document_data::after(k + 1)) + { + const char* const key = d.str(*k); + const std::uint64_t hash = key_hash(key, k->len); // (a cast of the call would be useless where std::uint64_t is std::size_t) + std::size_t i = static_cast(hash) & mask; + bool duplicate = false; + while (slots[i] != 0) + { + const node* const other = obj + slots[i]; + if (other->len == k->len && (k->len == 0 || std::memcmp(d.str(*other), key, k->len) == 0)) + { + duplicate = true; // keep the first + break; + } + i = (i + 1) & mask; + } + if (!duplicate) + { + slots[i] = static_cast(k - obj); + } + } + d.indexes.push_back(document_data::object_index{start, static_cast(mask)}); + obj->extra = static_cast(d.indexes.size()); +} + +/// build the tables of the large objects the parser noted +inline void build_object_indexes(document_data& d) +{ + for (const std::uint32_t i : d.large_objects) + { + build_object_index(d, d.tape + i); + } +} + +/// the key node of the first member with this key of an indexed object, or +/// nullptr +inline const node* find_indexed(const document_data& d, const node* obj, const char* key, std::size_t n) noexcept +{ + const document_data::object_index& ix = d.indexes[obj->extra - 1u]; + const std::uint32_t* const slots = d.index_slots.data() + ix.start; + const std::uint64_t hash = key_hash(key, n); // (a cast of the call would be useless where std::uint64_t is std::size_t) + std::size_t i = static_cast(hash) & ix.mask; + for (;;) + { + const std::uint32_t s = slots[i]; + if (s == 0) + { + return nullptr; + } + const node* const k = obj + s; + if (k->len == n && (n == 0 || std::memcmp(d.str(*k), key, n) == 0)) + { + return k; + } + i = (i + 1) & ix.mask; + } +} + +} // namespace view +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END diff --git a/include/nlohmann/detail/view/scan.hpp b/include/nlohmann/detail/view/scan.hpp index 6d732ae36..427a53a10 100644 --- a/include/nlohmann/detail/view/scan.hpp +++ b/include/nlohmann/detail/view/scan.hpp @@ -16,6 +16,7 @@ #include #include +#include // Scanning primitives of the view's parser. The unrolled checks at fixed // offsets follow yyjson (https://github.com/ibireme/yyjson, MIT license): the @@ -62,20 +63,43 @@ NLOHMANN_VIEW_ALWAYS_INLINE std::uint16_t load16(const unsigned char* p) noexcep /// Advance over plain string bytes and well-formed UTF-8. Stops at a quote, /// a backslash, a control character, ill-formed UTF-8, or the end. The first -/// 16 bytes are checked one by one, so that the position advances by -/// constants in predicted branches (most strings are short); longer runs -/// continue eight bytes at a time. +/// bytes are checked one by one, so that the position advances by constants +/// in predicted branches: 16 for keys, whose lengths repeat from record to +/// record, and 8 for string values (Value) where a vector loop follows, as +/// their lengths vary more. Longer runs continue 16 bytes at a time with NEON +/// or SSE2, else eight bytes at a time. With SSE2, the run is checked 16 bytes +/// at a time from its first byte instead: on x86-64, one compare that finds +/// the end of most keys and short values is faster than a branch per byte (on +/// AArch64, where a NEON mask costs more and branches predict well, slower). +template NLOHMANN_VIEW_ALWAYS_INLINE const unsigned char* scan_string_run(const unsigned char* p, const unsigned char* e) noexcept { const std::uint8_t* plain = string_plain(); for (;;) { +#if NLOHMANN_VIEW_SSE2 + p = vector_plain_run(p, e); +#else if (e - p >= 16) { #define NLOHMANN_VIEW_STEP(i) if (NLOHMANN_VIEW_LIKELY(plain[p[i]] != 0)) {} else { p += (i); goto stop; } - NLOHMANN_VIEW_REPEAT16(NLOHMANN_VIEW_STEP) + NLOHMANN_VIEW_STEP(0) NLOHMANN_VIEW_STEP(1) NLOHMANN_VIEW_STEP(2) NLOHMANN_VIEW_STEP(3) + NLOHMANN_VIEW_STEP(4) NLOHMANN_VIEW_STEP(5) NLOHMANN_VIEW_STEP(6) NLOHMANN_VIEW_STEP(7) + if (!Value || !NLOHMANN_VIEW_VECTOR) + { + NLOHMANN_VIEW_STEP(8) NLOHMANN_VIEW_STEP(9) NLOHMANN_VIEW_STEP(10) NLOHMANN_VIEW_STEP(11) + NLOHMANN_VIEW_STEP(12) NLOHMANN_VIEW_STEP(13) NLOHMANN_VIEW_STEP(14) NLOHMANN_VIEW_STEP(15) + p += 8; + } #undef NLOHMANN_VIEW_STEP - p += 16; + p += 8; +#if NLOHMANN_VIEW_VECTOR + p = vector_plain_run(p, e); + if (p != e && plain[*p] == 0) + { + goto stop; + } +#else while (e - p >= 8) { const std::uint64_t special = swar_string_special(read_eight_bytes(p)); @@ -86,8 +110,10 @@ NLOHMANN_VIEW_ALWAYS_INLINE const unsigned char* scan_string_run(const unsigned } p += 8; } +#endif continue; } +#endif while (p != e && plain[*p] != 0) { ++p; @@ -96,11 +122,23 @@ NLOHMANN_VIEW_ALWAYS_INLINE const unsigned char* scan_string_run(const unsigned { return p; } +#if !NLOHMANN_VIEW_SSE2 stop: +#endif if (*p < 0x80) { return p; // quote, backslash, or control character } +#if NLOHMANN_VIEW_VECTOR_UTF8 +#if NLOHMANN_VIEW_SSSE3_DISPATCH + if (NLOHMANN_VIEW_LIKELY(cpu_has_ssse3())) +#endif + { + // non-ASCII: the vector check, out of line + return scan_string_vector(p, e, plain); + } +#endif +#if !NLOHMANN_VIEW_VECTOR_UTF8 || NLOHMANN_VIEW_SSSE3_DISPATCH // non-ASCII: a run of well-formed sequences (the library's check, so // that exactly what json::parse accepts is accepted) do @@ -113,6 +151,7 @@ stop: p += n; } while (p != e && *p >= 0x80); +#endif } } diff --git a/include/nlohmann/detail/view/simd.hpp b/include/nlohmann/detail/view/simd.hpp new file mode 100644 index 000000000..f92b62466 --- /dev/null +++ b/include/nlohmann/detail/view/simd.hpp @@ -0,0 +1,335 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-FileCopyrightText: 2018-2025 The simdjson authors +// SPDX-License-Identifier: MIT + +#pragma once + +#include // array +#include // atomic +#include // size_t +#include // uint8_t, uint64_t + +#include +#include + +// Vector code for long runs of string bytes. NEON (AArch64) and SSE2 (x86-64) +// belong to the baseline instruction sets and are used by default. The vector +// UTF-8 check needs NEON or SSSE3. SSSE3 is not part of x86-64, and the code +// must not depend on the flags of a translation unit (two translation units +// with different flags would have different definitions of the same inline +// functions): the check is compiled for SSSE3 with a function attribute and +// used where the CPU has SSSE3 (all x86-64 CPUs since about 2011), else the +// portable check. JSON_VIEW_USE_SSSE3 skips the CPU check (for code compiled +// for SSSE3 anyway); JSON_VIEW_NO_SIMD selects the portable code. +#if !defined(JSON_VIEW_NO_SIMD) && defined(__aarch64__) && (defined(__GNUC__) || defined(__clang__)) && NLOHMANN_VIEW_LITTLE_ENDIAN + #include + #define NLOHMANN_VIEW_NEON 1 +#else + #define NLOHMANN_VIEW_NEON 0 +#endif +#if !defined(JSON_VIEW_NO_SIMD) && !NLOHMANN_VIEW_NEON && (defined(__SSE2__) || defined(_M_X64) || (defined(_M_IX86_FP) && _M_IX86_FP >= 2)) + #include + #define NLOHMANN_VIEW_SSE2 1 +#else + #define NLOHMANN_VIEW_SSE2 0 +#endif +#if NLOHMANN_VIEW_SSE2 && defined(JSON_VIEW_USE_SSSE3) + #include + #define NLOHMANN_VIEW_SSSE3 1 // NOLINT(cppcoreguidelines-macro-to-enum,modernize-macro-to-enum) +#else + #define NLOHMANN_VIEW_SSSE3 0 // NOLINT(cppcoreguidelines-macro-to-enum,modernize-macro-to-enum) +#endif +#if NLOHMANN_VIEW_SSE2 && !NLOHMANN_VIEW_SSSE3 && ((defined(__clang__) && __clang_major__ >= 4) || (defined(__GNUC__) && !defined(__clang__) && (__GNUC__ > 4 || (__GNUC__ == 4 && __GNUC_MINOR__ >= 9)))) + // (GCC before 4.9 has no SSSE3 intrinsics without -mssse3) + #include + #include + #define NLOHMANN_VIEW_SSSE3_DISPATCH 1 // NOLINT(cppcoreguidelines-macro-to-enum,modernize-macro-to-enum) + #define NLOHMANN_VIEW_SSSE3_TARGET __attribute__((target("ssse3"))) +#elif NLOHMANN_VIEW_SSE2 && !NLOHMANN_VIEW_SSSE3 && defined(_MSC_VER) + // (MSVC compiles intrinsics of any instruction set) + #include + #include + #define NLOHMANN_VIEW_SSSE3_DISPATCH 1 // NOLINT(cppcoreguidelines-macro-to-enum,modernize-macro-to-enum) + #define NLOHMANN_VIEW_SSSE3_TARGET +#else + #define NLOHMANN_VIEW_SSSE3_DISPATCH 0 // NOLINT(cppcoreguidelines-macro-to-enum,modernize-macro-to-enum) + #define NLOHMANN_VIEW_SSSE3_TARGET +#endif +#define NLOHMANN_VIEW_VECTOR (NLOHMANN_VIEW_NEON || NLOHMANN_VIEW_SSE2) +#define NLOHMANN_VIEW_VECTOR_UTF8 (NLOHMANN_VIEW_NEON || NLOHMANN_VIEW_SSSE3 || NLOHMANN_VIEW_SSSE3_DISPATCH) + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ +namespace view +{ + +#if NLOHMANN_VIEW_VECTOR +/*! +@brief the first byte of a string run that is a quote, a backslash, a control +character, or not ASCII, 16 bytes per step + +Stops at such a byte, or where fewer than 16 bytes are left (the caller tells +the two apart). A signed compare with 0x20 finds control characters and +non-ASCII bytes at once. +*/ +NLOHMANN_VIEW_ALWAYS_INLINE const unsigned char* vector_plain_run(const unsigned char* p, const unsigned char* e) noexcept +{ + while (e - p >= 16) + { +#if NLOHMANN_VIEW_NEON + const uint8x16_t in = vld1q_u8(p); + const uint8x16_t special = vorrq_u8(vorrq_u8(vceqq_u8(in, vdupq_n_u8('"')), vceqq_u8(in, vdupq_n_u8('\\'))), + vcltq_s8(vreinterpretq_s8_u8(in), vdupq_n_s8(0x20))); + // one nibble per byte (the usual NEON replacement of x86's movemask, see + // D. Kutenin, "Porting x86 vector bitmask optimizations to Arm NEON", 2022) + const std::uint64_t bits = vget_lane_u64(vreinterpret_u64_u8(vshrn_n_u16(vreinterpretq_u16_u8(special), 4)), 0); + if (bits != 0) + { + return p + (count_trailing_zeros(bits) >> 2u); + } +#else + const __m128i in = _mm_loadu_si128(static_cast(static_cast(p))); + const __m128i special = _mm_or_si128(_mm_or_si128(_mm_cmpeq_epi8(in, _mm_set1_epi8('"')), _mm_cmpeq_epi8(in, _mm_set1_epi8('\\'))), + _mm_cmplt_epi8(in, _mm_set1_epi8(0x20))); + const auto bits = static_cast(static_cast(_mm_movemask_epi8(special))); + if (bits != 0) + { + return p + count_trailing_zeros(bits); + } +#endif + p += 16; + } + return p; +} +#endif + +#if NLOHMANN_VIEW_SSSE3_DISPATCH +/// whether the CPU has SSSE3 (CPUID leaf 1, ECX bit 9) +inline bool cpu_ssse3() noexcept +{ +#if defined(_MSC_VER) && !defined(__clang__) + std::array regs {{}}; + __cpuid(regs.data(), 1); + return (static_cast(regs[2]) & (1u << 9u)) != 0; +#else + unsigned eax = 0; + unsigned ebx = 0; + unsigned ecx = 0; + unsigned edx = 0; + return __get_cpuid(1, &eax, &ebx, &ecx, &edx) != 0 && (ecx & (1u << 9u)) != 0; +#endif +} + +/// whether the CPU has SSSE3, asked once: the answer is kept in an atomic +/// that is initialized at compile time, so that neither a guard of a local +/// static nor a global constructor is needed (threads that ask at the same +/// time all store the same answer) +NLOHMANN_VIEW_ALWAYS_INLINE bool cpu_has_ssse3() noexcept +{ + static std::atomic known{0}; // 0: not asked yet, 1: no, 2: yes + int state = known.load(std::memory_order_relaxed); + if (NLOHMANN_VIEW_UNLIKELY(state == 0)) + { + state = cpu_ssse3() ? 2 : 1; + known.store(state, std::memory_order_relaxed); + } + return state == 2; +} +#endif + +#if NLOHMANN_VIEW_VECTOR_UTF8 +/// Tables of the UTF-8 check of J. Keiser and D. Lemire, "Validating UTF-8 In +/// Less Than One Instruction Per Byte" (2021), as in simdjson ("lookup4"): each +/// maps a nibble (high and low nibble of the previous byte, high nibble of the +/// current byte) to the errors it allows; a byte pair is ill-formed if all +/// three have an error bit in common. +template +struct utf8_lookup4 +{ + static constexpr std::uint8_t too_short = 1u << 0u, too_long = 1u << 1u, overlong_3 = 1u << 2u, too_large = 1u << 3u; + static constexpr std::uint8_t surrogate = 1u << 4u, overlong_2 = 1u << 5u, too_large_1000 = 1u << 6u, overlong_4 = 1u << 6u; + static constexpr std::uint8_t two_conts = 1u << 7u, carry = too_short | too_long | two_conts; + static const std::array byte_1_high; + static const std::array byte_1_low; + static const std::array byte_2_high; +}; + +template +const std::array utf8_lookup4::byte_1_high = +{ + { + too_long, too_long, too_long, too_long, too_long, too_long, too_long, too_long, + two_conts, two_conts, two_conts, two_conts, + too_short | overlong_2, too_short, too_short | overlong_3 | surrogate, too_short | too_large | too_large_1000 | overlong_4 + } +}; + +template +const std::array utf8_lookup4::byte_1_low = +{ + { + carry | overlong_3 | overlong_2 | overlong_4, carry | overlong_2, carry, carry, + carry | too_large, carry | too_large | too_large_1000, carry | too_large | too_large_1000, carry | too_large | too_large_1000, + carry | too_large | too_large_1000, carry | too_large | too_large_1000, carry | too_large | too_large_1000, carry | too_large | too_large_1000, + carry | too_large | too_large_1000, carry | too_large | too_large_1000 | surrogate, carry | too_large | too_large_1000, carry | too_large | too_large_1000 + } +}; + +template +const std::array utf8_lookup4::byte_2_high = +{ + { + too_short, too_short, too_short, too_short, too_short, too_short, too_short, too_short, + static_cast(too_long | overlong_2 | two_conts | overlong_3 | too_large_1000 | overlong_4), + static_cast(too_long | overlong_2 | two_conts | overlong_3 | too_large), + static_cast(too_long | overlong_2 | two_conts | surrogate | too_large), + static_cast(too_long | overlong_2 | two_conts | surrogate | too_large), + too_short, too_short, too_short, too_short + } +}; + +/// the end of scan_string_vector from block, where the vector loop stopped +/// (ill-formed UTF-8, or fewer than 16 bytes left): one byte or sequence at a +/// time, from the start of a sequence that crosses into the block +inline const unsigned char* scan_string_finish(const unsigned char* p, const unsigned char* block, const unsigned char* e, const std::uint8_t* plain) noexcept +{ + for (int i = 1; i <= 3 && block - i >= p; ++i) + { + const unsigned char c = block[-i]; + if (c < 0x80) + { + break; + } + if (c >= 0xC0) + { + const int len = 2 + static_cast(c >= 0xE0) + static_cast(c >= 0xF0); + if (len > i) + { + block -= i; + } + break; + } + } + for (p = block; p != e;) + { + if (*p < 0x80) + { + if (plain[*p] == 0) + { + return p; + } + ++p; + continue; + } + const std::size_t n = validate_one_utf8(p, static_cast(e - p)); + if (n == 0) + { + return p; + } + p += n; + } + return p; +} + +/*! +@brief the rest of a string from p (a character boundary), 16 bytes per step + +The first quote, backslash, or control character is found with vector +compares, and the UTF-8 check covers the bytes up to it. Returns where the +string scan stops, like scan_string_run: before ill-formed UTF-8 and for the +last bytes of the input, the bytes are checked one sequence at a time. Out of +line, so that no constants of the check occupy registers in the parse loop. +On x86-64, it is compiled for SSSE3 (see cpu_has_ssse3()). +*/ +NLOHMANN_VIEW_SSSE3_TARGET NLOHMANN_VIEW_NOINLINE inline const unsigned char* scan_string_vector(const unsigned char* p, const unsigned char* e, const std::uint8_t* plain) noexcept +{ + using lookup = utf8_lookup4<>; + const unsigned char* block = p; +#if NLOHMANN_VIEW_NEON + const uint8x16_t t1h = vld1q_u8(lookup::byte_1_high.data()); + const uint8x16_t t1l = vld1q_u8(lookup::byte_1_low.data()); + const uint8x16_t t2h = vld1q_u8(lookup::byte_2_high.data()); + uint8x16_t prev = vdupq_n_u8(0); + while (e - block >= 16) + { + const uint8x16_t in = vld1q_u8(block); + const uint8x16_t special = vorrq_u8(vorrq_u8(vceqq_u8(in, vdupq_n_u8('"')), vceqq_u8(in, vdupq_n_u8('\\'))), vcltq_u8(in, vdupq_n_u8(0x20))); + const uint8x16_t prev1 = vextq_u8(prev, in, 15); + const uint8x16_t sc = vandq_u8(vandq_u8(vqtbl1q_u8(t1h, vshrq_n_u8(prev1, 4)), vqtbl1q_u8(t1l, vandq_u8(prev1, vdupq_n_u8(0x0F)))), vqtbl1q_u8(t2h, vshrq_n_u8(in, 4))); + const uint8x16_t must23 = vorrq_u8(vqsubq_u8(vextq_u8(prev, in, 14), vdupq_n_u8(0xE0 - 0x80)), vqsubq_u8(vextq_u8(prev, in, 13), vdupq_n_u8(0xF0 - 0x80))); + const uint8x16_t err = veorq_u8(vandq_u8(must23, vdupq_n_u8(0x80)), sc); + const std::uint64_t special_bits = vget_lane_u64(vreinterpret_u64_u8(vshrn_n_u16(vreinterpretq_u16_u8(special), 4)), 0); + const std::uint64_t err_bits = vget_lane_u64(vreinterpret_u64_u8(vshrn_n_u16(vreinterpretq_u16_u8(vtstq_u8(err, err)), 4)), 0); + if (special_bits != 0) + { + // errors up to the special byte count (an incomplete sequence + // before a quote shows at the quote); the bytes after it do not + const unsigned k = static_cast(count_trailing_zeros(special_bits)) >> 2u; + const std::uint64_t upto = k == 15 ? ~std::uint64_t{0} : + (std::uint64_t{1} << (4u * (k + 1u))) - 1u; + if ((err_bits & upto) == 0) + { + return block + k; + } + break; + } + if (err_bits != 0) + { + break; + } + prev = in; + block += 16; + } +#else + // the same with SSSE3 (pshufb for the table lookups; nibbles from 16-bit + // shifts, as there are no byte shifts) + const __m128i t1h = _mm_loadu_si128(static_cast(static_cast(lookup::byte_1_high.data()))); + const __m128i t1l = _mm_loadu_si128(static_cast(static_cast(lookup::byte_1_low.data()))); + const __m128i t2h = _mm_loadu_si128(static_cast(static_cast(lookup::byte_2_high.data()))); + const __m128i nibble = _mm_set1_epi8(0x0F); + const __m128i zero = _mm_setzero_si128(); + __m128i prev = zero; + while (e - block >= 16) + { + const __m128i in = _mm_loadu_si128(static_cast(static_cast(block))); + const __m128i special = _mm_or_si128(_mm_or_si128(_mm_cmpeq_epi8(in, _mm_set1_epi8('"')), _mm_cmpeq_epi8(in, _mm_set1_epi8('\\'))), + _mm_cmpeq_epi8(_mm_subs_epu8(in, _mm_set1_epi8(0x1F)), zero)); // in < 0x20 + const __m128i prev1 = _mm_alignr_epi8(in, prev, 15); + const __m128i sc = _mm_and_si128(_mm_and_si128(_mm_shuffle_epi8(t1h, _mm_and_si128(_mm_srli_epi16(prev1, 4), nibble)), + _mm_shuffle_epi8(t1l, _mm_and_si128(prev1, nibble))), + _mm_shuffle_epi8(t2h, _mm_and_si128(_mm_srli_epi16(in, 4), nibble))); + const __m128i must23 = _mm_or_si128(_mm_subs_epu8(_mm_alignr_epi8(in, prev, 14), _mm_set1_epi8(0xE0 - 0x80)), + _mm_subs_epu8(_mm_alignr_epi8(in, prev, 13), _mm_set1_epi8(0xF0 - 0x80))); + const __m128i err = _mm_xor_si128(_mm_and_si128(must23, _mm_set1_epi8(static_cast(-128))), sc); + const auto special_bits = static_cast(_mm_movemask_epi8(special)); + const auto err_bits = ~static_cast(_mm_movemask_epi8(_mm_cmpeq_epi8(err, zero))) & 0xFFFFu; + if (special_bits != 0) + { + const unsigned k = static_cast(count_trailing_zeros(static_cast(special_bits))); + if ((err_bits & ((2u << k) - 1u)) == 0) + { + return block + k; + } + break; + } + if (err_bits != 0) + { + break; + } + prev = in; + block += 16; + } +#endif + return scan_string_finish(p, block, e, plain); +} +#endif + +} // namespace view +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END diff --git a/include/nlohmann/json_view.hpp b/include/nlohmann/json_view.hpp index 882101336..d70bde43f 100644 --- a/include/nlohmann/json_view.hpp +++ b/include/nlohmann/json_view.hpp @@ -25,6 +25,7 @@ #define INCLUDE_NLOHMANN_JSON_VIEW_HPP_ #include // size_t +#include // uint32_t #include // memcpy, strlen #include // distance, input_iterator_tag, iterator_traits #include // map @@ -56,6 +57,7 @@ #include #include #include +#include #include #include #include @@ -926,7 +928,9 @@ class basic_json_document } return sizeof(document_data) + (m_data->inline_cap * sizeof(detail::view::node)) + (m_data->tape != m_data->inline_tape ? m_data->tape_cap * sizeof(detail::view::node) : 0) - + m_data->arena.capacity() + m_data->owned.capacity(); + + m_data->arena.capacity() + m_data->owned.capacity() + + (m_data->indexes.capacity() * sizeof(document_data::object_index)) + (m_data->index_slots.capacity() * sizeof(std::uint32_t)) + + (m_data->large_objects.capacity() * sizeof(std::uint32_t)); } /// release unused capacity of the index and the decoded strings; like @@ -996,6 +1000,9 @@ class basic_json_document d.size = size; d.tape_size = 0; d.arena.clear(); + d.indexes.clear(); + d.index_slots.clear(); + d.large_objects.clear(); d.discarded = true; detail::view::parse_failure failure; bool ok = false; @@ -1011,6 +1018,7 @@ class basic_json_document { d.base[0] = d.src; d.base[1] = d.arena.data(); + detail::view::build_object_indexes(d); d.discarded = false; return; } diff --git a/single_include/nlohmann/json_view.hpp b/single_include/nlohmann/json_view.hpp index cdec6523d..5e69bd9bd 100644 --- a/single_include/nlohmann/json_view.hpp +++ b/single_include/nlohmann/json_view.hpp @@ -25,6 +25,7 @@ #define INCLUDE_NLOHMANN_JSON_VIEW_HPP_ #include // size_t +#include // uint32_t #include // memcpy, strlen #include // distance, input_iterator_tag, iterator_traits #include // map @@ -81,9 +82,11 @@ #include // array #include // size_t +#include // uint32_t #include // memcpy #include // operator new, placement new #include // string +#include // vector // #include // #include @@ -279,6 +282,17 @@ struct document_data std::size_t inline_cap = 0; std::string arena{}; ///< decoded strings that contained escapes // NOLINT(readability-redundant-member-init) std::string owned{}; ///< owned copy of the input, if any // NOLINT(readability-redundant-member-init) + + // hash indexes of large objects (see object_index.hpp) + static constexpr std::uint32_t index_min_members = 128; + struct object_index + { + std::size_t start; ///< first slot in index_slots + std::uint32_t mask; ///< slot count - 1 (a power of two minus one) + }; + std::vector indexes{}; // NOLINT(readability-redundant-member-init) + std::vector index_slots{}; // NOLINT(readability-redundant-member-init) + std::vector large_objects{}; ///< positions of the objects to index (noted while parsing) // NOLINT(readability-redundant-member-init) std::array base = {{nullptr, nullptr, nullptr, nullptr}}; ///< string bases: source, arena (indexed by flags & node_flags::storage) bool discarded = true; @@ -306,7 +320,7 @@ struct document_data } }; - document_data() noexcept = default; + document_data() = default; document_data(const document_data&) = delete; document_data(document_data&&) = delete; document_data& operator=(const document_data&) = delete; @@ -395,6 +409,344 @@ NLOHMANN_JSON_NAMESPACE_END // #include // #include +// #include +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-FileCopyrightText: 2018-2025 The simdjson authors +// SPDX-License-Identifier: MIT + + + +#include // array +#include // atomic +#include // size_t +#include // uint8_t, uint64_t + +// #include +// #include + + +// Vector code for long runs of string bytes. NEON (AArch64) and SSE2 (x86-64) +// belong to the baseline instruction sets and are used by default. The vector +// UTF-8 check needs NEON or SSSE3. SSSE3 is not part of x86-64, and the code +// must not depend on the flags of a translation unit (two translation units +// with different flags would have different definitions of the same inline +// functions): the check is compiled for SSSE3 with a function attribute and +// used where the CPU has SSSE3 (all x86-64 CPUs since about 2011), else the +// portable check. JSON_VIEW_USE_SSSE3 skips the CPU check (for code compiled +// for SSSE3 anyway); JSON_VIEW_NO_SIMD selects the portable code. +#if !defined(JSON_VIEW_NO_SIMD) && defined(__aarch64__) && (defined(__GNUC__) || defined(__clang__)) && NLOHMANN_VIEW_LITTLE_ENDIAN + #include + #define NLOHMANN_VIEW_NEON 1 +#else + #define NLOHMANN_VIEW_NEON 0 +#endif +#if !defined(JSON_VIEW_NO_SIMD) && !NLOHMANN_VIEW_NEON && (defined(__SSE2__) || defined(_M_X64) || (defined(_M_IX86_FP) && _M_IX86_FP >= 2)) + #include + #define NLOHMANN_VIEW_SSE2 1 +#else + #define NLOHMANN_VIEW_SSE2 0 +#endif +#if NLOHMANN_VIEW_SSE2 && defined(JSON_VIEW_USE_SSSE3) + #include + #define NLOHMANN_VIEW_SSSE3 1 // NOLINT(cppcoreguidelines-macro-to-enum,modernize-macro-to-enum) +#else + #define NLOHMANN_VIEW_SSSE3 0 // NOLINT(cppcoreguidelines-macro-to-enum,modernize-macro-to-enum) +#endif +#if NLOHMANN_VIEW_SSE2 && !NLOHMANN_VIEW_SSSE3 && ((defined(__clang__) && __clang_major__ >= 4) || (defined(__GNUC__) && !defined(__clang__) && (__GNUC__ > 4 || (__GNUC__ == 4 && __GNUC_MINOR__ >= 9)))) + // (GCC before 4.9 has no SSSE3 intrinsics without -mssse3) + #include + #include + #define NLOHMANN_VIEW_SSSE3_DISPATCH 1 // NOLINT(cppcoreguidelines-macro-to-enum,modernize-macro-to-enum) + #define NLOHMANN_VIEW_SSSE3_TARGET __attribute__((target("ssse3"))) +#elif NLOHMANN_VIEW_SSE2 && !NLOHMANN_VIEW_SSSE3 && defined(_MSC_VER) + // (MSVC compiles intrinsics of any instruction set) + #include + #include + #define NLOHMANN_VIEW_SSSE3_DISPATCH 1 // NOLINT(cppcoreguidelines-macro-to-enum,modernize-macro-to-enum) + #define NLOHMANN_VIEW_SSSE3_TARGET +#else + #define NLOHMANN_VIEW_SSSE3_DISPATCH 0 // NOLINT(cppcoreguidelines-macro-to-enum,modernize-macro-to-enum) + #define NLOHMANN_VIEW_SSSE3_TARGET +#endif +#define NLOHMANN_VIEW_VECTOR (NLOHMANN_VIEW_NEON || NLOHMANN_VIEW_SSE2) +#define NLOHMANN_VIEW_VECTOR_UTF8 (NLOHMANN_VIEW_NEON || NLOHMANN_VIEW_SSSE3 || NLOHMANN_VIEW_SSSE3_DISPATCH) + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ +namespace view +{ + +#if NLOHMANN_VIEW_VECTOR +/*! +@brief the first byte of a string run that is a quote, a backslash, a control +character, or not ASCII, 16 bytes per step + +Stops at such a byte, or where fewer than 16 bytes are left (the caller tells +the two apart). A signed compare with 0x20 finds control characters and +non-ASCII bytes at once. +*/ +NLOHMANN_VIEW_ALWAYS_INLINE const unsigned char* vector_plain_run(const unsigned char* p, const unsigned char* e) noexcept +{ + while (e - p >= 16) + { +#if NLOHMANN_VIEW_NEON + const uint8x16_t in = vld1q_u8(p); + const uint8x16_t special = vorrq_u8(vorrq_u8(vceqq_u8(in, vdupq_n_u8('"')), vceqq_u8(in, vdupq_n_u8('\\'))), + vcltq_s8(vreinterpretq_s8_u8(in), vdupq_n_s8(0x20))); + // one nibble per byte (the usual NEON replacement of x86's movemask, see + // D. Kutenin, "Porting x86 vector bitmask optimizations to Arm NEON", 2022) + const std::uint64_t bits = vget_lane_u64(vreinterpret_u64_u8(vshrn_n_u16(vreinterpretq_u16_u8(special), 4)), 0); + if (bits != 0) + { + return p + (count_trailing_zeros(bits) >> 2u); + } +#else + const __m128i in = _mm_loadu_si128(static_cast(static_cast(p))); + const __m128i special = _mm_or_si128(_mm_or_si128(_mm_cmpeq_epi8(in, _mm_set1_epi8('"')), _mm_cmpeq_epi8(in, _mm_set1_epi8('\\'))), + _mm_cmplt_epi8(in, _mm_set1_epi8(0x20))); + const auto bits = static_cast(static_cast(_mm_movemask_epi8(special))); + if (bits != 0) + { + return p + count_trailing_zeros(bits); + } +#endif + p += 16; + } + return p; +} +#endif + +#if NLOHMANN_VIEW_SSSE3_DISPATCH +/// whether the CPU has SSSE3 (CPUID leaf 1, ECX bit 9) +inline bool cpu_ssse3() noexcept +{ +#if defined(_MSC_VER) && !defined(__clang__) + std::array regs {{}}; + __cpuid(regs.data(), 1); + return (static_cast(regs[2]) & (1u << 9u)) != 0; +#else + unsigned eax = 0; + unsigned ebx = 0; + unsigned ecx = 0; + unsigned edx = 0; + return __get_cpuid(1, &eax, &ebx, &ecx, &edx) != 0 && (ecx & (1u << 9u)) != 0; +#endif +} + +/// whether the CPU has SSSE3, asked once: the answer is kept in an atomic +/// that is initialized at compile time, so that neither a guard of a local +/// static nor a global constructor is needed (threads that ask at the same +/// time all store the same answer) +NLOHMANN_VIEW_ALWAYS_INLINE bool cpu_has_ssse3() noexcept +{ + static std::atomic known{0}; // 0: not asked yet, 1: no, 2: yes + int state = known.load(std::memory_order_relaxed); + if (NLOHMANN_VIEW_UNLIKELY(state == 0)) + { + state = cpu_ssse3() ? 2 : 1; + known.store(state, std::memory_order_relaxed); + } + return state == 2; +} +#endif + +#if NLOHMANN_VIEW_VECTOR_UTF8 +/// Tables of the UTF-8 check of J. Keiser and D. Lemire, "Validating UTF-8 In +/// Less Than One Instruction Per Byte" (2021), as in simdjson ("lookup4"): each +/// maps a nibble (high and low nibble of the previous byte, high nibble of the +/// current byte) to the errors it allows; a byte pair is ill-formed if all +/// three have an error bit in common. +template +struct utf8_lookup4 +{ + static constexpr std::uint8_t too_short = 1u << 0u, too_long = 1u << 1u, overlong_3 = 1u << 2u, too_large = 1u << 3u; + static constexpr std::uint8_t surrogate = 1u << 4u, overlong_2 = 1u << 5u, too_large_1000 = 1u << 6u, overlong_4 = 1u << 6u; + static constexpr std::uint8_t two_conts = 1u << 7u, carry = too_short | too_long | two_conts; + static const std::array byte_1_high; + static const std::array byte_1_low; + static const std::array byte_2_high; +}; + +template +const std::array utf8_lookup4::byte_1_high = +{ + { + too_long, too_long, too_long, too_long, too_long, too_long, too_long, too_long, + two_conts, two_conts, two_conts, two_conts, + too_short | overlong_2, too_short, too_short | overlong_3 | surrogate, too_short | too_large | too_large_1000 | overlong_4 + } +}; + +template +const std::array utf8_lookup4::byte_1_low = +{ + { + carry | overlong_3 | overlong_2 | overlong_4, carry | overlong_2, carry, carry, + carry | too_large, carry | too_large | too_large_1000, carry | too_large | too_large_1000, carry | too_large | too_large_1000, + carry | too_large | too_large_1000, carry | too_large | too_large_1000, carry | too_large | too_large_1000, carry | too_large | too_large_1000, + carry | too_large | too_large_1000, carry | too_large | too_large_1000 | surrogate, carry | too_large | too_large_1000, carry | too_large | too_large_1000 + } +}; + +template +const std::array utf8_lookup4::byte_2_high = +{ + { + too_short, too_short, too_short, too_short, too_short, too_short, too_short, too_short, + static_cast(too_long | overlong_2 | two_conts | overlong_3 | too_large_1000 | overlong_4), + static_cast(too_long | overlong_2 | two_conts | overlong_3 | too_large), + static_cast(too_long | overlong_2 | two_conts | surrogate | too_large), + static_cast(too_long | overlong_2 | two_conts | surrogate | too_large), + too_short, too_short, too_short, too_short + } +}; + +/// the end of scan_string_vector from block, where the vector loop stopped +/// (ill-formed UTF-8, or fewer than 16 bytes left): one byte or sequence at a +/// time, from the start of a sequence that crosses into the block +inline const unsigned char* scan_string_finish(const unsigned char* p, const unsigned char* block, const unsigned char* e, const std::uint8_t* plain) noexcept +{ + for (int i = 1; i <= 3 && block - i >= p; ++i) + { + const unsigned char c = block[-i]; + if (c < 0x80) + { + break; + } + if (c >= 0xC0) + { + const int len = 2 + static_cast(c >= 0xE0) + static_cast(c >= 0xF0); + if (len > i) + { + block -= i; + } + break; + } + } + for (p = block; p != e;) + { + if (*p < 0x80) + { + if (plain[*p] == 0) + { + return p; + } + ++p; + continue; + } + const std::size_t n = validate_one_utf8(p, static_cast(e - p)); + if (n == 0) + { + return p; + } + p += n; + } + return p; +} + +/*! +@brief the rest of a string from p (a character boundary), 16 bytes per step + +The first quote, backslash, or control character is found with vector +compares, and the UTF-8 check covers the bytes up to it. Returns where the +string scan stops, like scan_string_run: before ill-formed UTF-8 and for the +last bytes of the input, the bytes are checked one sequence at a time. Out of +line, so that no constants of the check occupy registers in the parse loop. +On x86-64, it is compiled for SSSE3 (see cpu_has_ssse3()). +*/ +NLOHMANN_VIEW_SSSE3_TARGET NLOHMANN_VIEW_NOINLINE inline const unsigned char* scan_string_vector(const unsigned char* p, const unsigned char* e, const std::uint8_t* plain) noexcept +{ + using lookup = utf8_lookup4<>; + const unsigned char* block = p; +#if NLOHMANN_VIEW_NEON + const uint8x16_t t1h = vld1q_u8(lookup::byte_1_high.data()); + const uint8x16_t t1l = vld1q_u8(lookup::byte_1_low.data()); + const uint8x16_t t2h = vld1q_u8(lookup::byte_2_high.data()); + uint8x16_t prev = vdupq_n_u8(0); + while (e - block >= 16) + { + const uint8x16_t in = vld1q_u8(block); + const uint8x16_t special = vorrq_u8(vorrq_u8(vceqq_u8(in, vdupq_n_u8('"')), vceqq_u8(in, vdupq_n_u8('\\'))), vcltq_u8(in, vdupq_n_u8(0x20))); + const uint8x16_t prev1 = vextq_u8(prev, in, 15); + const uint8x16_t sc = vandq_u8(vandq_u8(vqtbl1q_u8(t1h, vshrq_n_u8(prev1, 4)), vqtbl1q_u8(t1l, vandq_u8(prev1, vdupq_n_u8(0x0F)))), vqtbl1q_u8(t2h, vshrq_n_u8(in, 4))); + const uint8x16_t must23 = vorrq_u8(vqsubq_u8(vextq_u8(prev, in, 14), vdupq_n_u8(0xE0 - 0x80)), vqsubq_u8(vextq_u8(prev, in, 13), vdupq_n_u8(0xF0 - 0x80))); + const uint8x16_t err = veorq_u8(vandq_u8(must23, vdupq_n_u8(0x80)), sc); + const std::uint64_t special_bits = vget_lane_u64(vreinterpret_u64_u8(vshrn_n_u16(vreinterpretq_u16_u8(special), 4)), 0); + const std::uint64_t err_bits = vget_lane_u64(vreinterpret_u64_u8(vshrn_n_u16(vreinterpretq_u16_u8(vtstq_u8(err, err)), 4)), 0); + if (special_bits != 0) + { + // errors up to the special byte count (an incomplete sequence + // before a quote shows at the quote); the bytes after it do not + const unsigned k = static_cast(count_trailing_zeros(special_bits)) >> 2u; + const std::uint64_t upto = k == 15 ? ~std::uint64_t{0} : + (std::uint64_t{1} << (4u * (k + 1u))) - 1u; + if ((err_bits & upto) == 0) + { + return block + k; + } + break; + } + if (err_bits != 0) + { + break; + } + prev = in; + block += 16; + } +#else + // the same with SSSE3 (pshufb for the table lookups; nibbles from 16-bit + // shifts, as there are no byte shifts) + const __m128i t1h = _mm_loadu_si128(static_cast(static_cast(lookup::byte_1_high.data()))); + const __m128i t1l = _mm_loadu_si128(static_cast(static_cast(lookup::byte_1_low.data()))); + const __m128i t2h = _mm_loadu_si128(static_cast(static_cast(lookup::byte_2_high.data()))); + const __m128i nibble = _mm_set1_epi8(0x0F); + const __m128i zero = _mm_setzero_si128(); + __m128i prev = zero; + while (e - block >= 16) + { + const __m128i in = _mm_loadu_si128(static_cast(static_cast(block))); + const __m128i special = _mm_or_si128(_mm_or_si128(_mm_cmpeq_epi8(in, _mm_set1_epi8('"')), _mm_cmpeq_epi8(in, _mm_set1_epi8('\\'))), + _mm_cmpeq_epi8(_mm_subs_epu8(in, _mm_set1_epi8(0x1F)), zero)); // in < 0x20 + const __m128i prev1 = _mm_alignr_epi8(in, prev, 15); + const __m128i sc = _mm_and_si128(_mm_and_si128(_mm_shuffle_epi8(t1h, _mm_and_si128(_mm_srli_epi16(prev1, 4), nibble)), + _mm_shuffle_epi8(t1l, _mm_and_si128(prev1, nibble))), + _mm_shuffle_epi8(t2h, _mm_and_si128(_mm_srli_epi16(in, 4), nibble))); + const __m128i must23 = _mm_or_si128(_mm_subs_epu8(_mm_alignr_epi8(in, prev, 14), _mm_set1_epi8(0xE0 - 0x80)), + _mm_subs_epu8(_mm_alignr_epi8(in, prev, 13), _mm_set1_epi8(0xF0 - 0x80))); + const __m128i err = _mm_xor_si128(_mm_and_si128(must23, _mm_set1_epi8(static_cast(-128))), sc); + const auto special_bits = static_cast(_mm_movemask_epi8(special)); + const auto err_bits = ~static_cast(_mm_movemask_epi8(_mm_cmpeq_epi8(err, zero))) & 0xFFFFu; + if (special_bits != 0) + { + const unsigned k = static_cast(count_trailing_zeros(static_cast(special_bits))); + if ((err_bits & ((2u << k) - 1u)) == 0) + { + return block + k; + } + break; + } + if (err_bits != 0) + { + break; + } + prev = in; + block += 16; + } +#endif + return scan_string_finish(p, block, e, plain); +} +#endif + +} // namespace view +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END + // Scanning primitives of the view's parser. The unrolled checks at fixed // offsets follow yyjson (https://github.com/ibireme/yyjson, MIT license): the @@ -441,20 +793,43 @@ NLOHMANN_VIEW_ALWAYS_INLINE std::uint16_t load16(const unsigned char* p) noexcep /// Advance over plain string bytes and well-formed UTF-8. Stops at a quote, /// a backslash, a control character, ill-formed UTF-8, or the end. The first -/// 16 bytes are checked one by one, so that the position advances by -/// constants in predicted branches (most strings are short); longer runs -/// continue eight bytes at a time. +/// bytes are checked one by one, so that the position advances by constants +/// in predicted branches: 16 for keys, whose lengths repeat from record to +/// record, and 8 for string values (Value) where a vector loop follows, as +/// their lengths vary more. Longer runs continue 16 bytes at a time with NEON +/// or SSE2, else eight bytes at a time. With SSE2, the run is checked 16 bytes +/// at a time from its first byte instead: on x86-64, one compare that finds +/// the end of most keys and short values is faster than a branch per byte (on +/// AArch64, where a NEON mask costs more and branches predict well, slower). +template NLOHMANN_VIEW_ALWAYS_INLINE const unsigned char* scan_string_run(const unsigned char* p, const unsigned char* e) noexcept { const std::uint8_t* plain = string_plain(); for (;;) { +#if NLOHMANN_VIEW_SSE2 + p = vector_plain_run(p, e); +#else if (e - p >= 16) { #define NLOHMANN_VIEW_STEP(i) if (NLOHMANN_VIEW_LIKELY(plain[p[i]] != 0)) {} else { p += (i); goto stop; } - NLOHMANN_VIEW_REPEAT16(NLOHMANN_VIEW_STEP) + NLOHMANN_VIEW_STEP(0) NLOHMANN_VIEW_STEP(1) NLOHMANN_VIEW_STEP(2) NLOHMANN_VIEW_STEP(3) + NLOHMANN_VIEW_STEP(4) NLOHMANN_VIEW_STEP(5) NLOHMANN_VIEW_STEP(6) NLOHMANN_VIEW_STEP(7) + if (!Value || !NLOHMANN_VIEW_VECTOR) + { + NLOHMANN_VIEW_STEP(8) NLOHMANN_VIEW_STEP(9) NLOHMANN_VIEW_STEP(10) NLOHMANN_VIEW_STEP(11) + NLOHMANN_VIEW_STEP(12) NLOHMANN_VIEW_STEP(13) NLOHMANN_VIEW_STEP(14) NLOHMANN_VIEW_STEP(15) + p += 8; + } #undef NLOHMANN_VIEW_STEP - p += 16; + p += 8; +#if NLOHMANN_VIEW_VECTOR + p = vector_plain_run(p, e); + if (p != e && plain[*p] == 0) + { + goto stop; + } +#else while (e - p >= 8) { const std::uint64_t special = swar_string_special(read_eight_bytes(p)); @@ -465,8 +840,10 @@ NLOHMANN_VIEW_ALWAYS_INLINE const unsigned char* scan_string_run(const unsigned } p += 8; } +#endif continue; } +#endif while (p != e && plain[*p] != 0) { ++p; @@ -475,11 +852,23 @@ NLOHMANN_VIEW_ALWAYS_INLINE const unsigned char* scan_string_run(const unsigned { return p; } +#if !NLOHMANN_VIEW_SSE2 stop: +#endif if (*p < 0x80) { return p; // quote, backslash, or control character } +#if NLOHMANN_VIEW_VECTOR_UTF8 +#if NLOHMANN_VIEW_SSSE3_DISPATCH + if (NLOHMANN_VIEW_LIKELY(cpu_has_ssse3())) +#endif + { + // non-ASCII: the vector check, out of line + return scan_string_vector(p, e, plain); + } +#endif +#if !NLOHMANN_VIEW_VECTOR_UTF8 || NLOHMANN_VIEW_SSSE3_DISPATCH // non-ASCII: a run of well-formed sequences (the library's check, so // that exactly what json::parse accepts is accepted) do @@ -492,6 +881,7 @@ stop: p += n; } while (p != e && *p >= 0x80); +#endif } } @@ -660,6 +1050,13 @@ class builder frame shallow[64]; // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays): not initialized on purpose; filled as containers open std::vector deep{}; + /// remember an object to index after parsing (out of line, so that the + /// parse loop only has a call for it) + NLOHMANN_VIEW_NOINLINE void note_large_object(std::uint32_t idx) + { + doc.large_objects.push_back(idx); + } + NLOHMANN_VIEW_NOINLINE bool fail(error_code c, const unsigned char* at) noexcept { m_failure.code = c; @@ -1045,7 +1442,7 @@ class builder switch (cur()) \ { \ case '"': \ - if (NLOHMANN_VIEW_UNLIKELY(!string())) { return false; } \ + if (NLOHMANN_VIEW_UNLIKELY(!string())) { return false; } \ goto NEXT; \ case '{': \ open(value_t::object); \ @@ -1129,7 +1526,7 @@ obj_key: { return fail(error_code::expected_key); } - if (NLOHMANN_VIEW_UNLIKELY(!string())) + if (NLOHMANN_VIEW_UNLIKELY(!string())) { return false; } @@ -1170,19 +1567,27 @@ obj_next: if (enabled(TrailingCommas) && cur() == '}') { ++p; - goto close_container; + goto close_object; } goto obj_key; } if (cur() == '}') { ++p; - goto close_container; + goto close_object; } return fail(error_code::expected_object_end); #undef NLOHMANN_VIEW_VALUE +close_object: + // a large object gets a hash index (objects only, so that closing + // an array pays nothing for this) + if (NLOHMANN_VIEW_UNLIKELY(cur_count >= document_data::index_min_members)) + { + cold.note_large_object(cur_idx); + } + close_container: close(); if (NLOHMANN_VIEW_UNLIKELY(depth == 0)) @@ -1232,7 +1637,7 @@ root_done: switch (cur()) { case '"': - return string(); + return string(); case 't': return literal("true", 4, value_t::boolean, node_flags::is_true); case 'f': @@ -1355,14 +1760,19 @@ indent_done: const auto idx = static_cast(emit(k, 0, 0, static_cast(p - b), 0) - base); if (depth != 0) { - const frame f = {cur_idx, cur_count, cur_is_object}; if (NLOHMANN_VIEW_LIKELY(depth <= 64)) { - cold.shallow[depth - 1] = f; + // field by field: a frame put together on the stack and + // copied would be read back wider than it was written, + // and that load waits until the stores are done + frame& f = cold.shallow[depth - 1]; + f.idx = cur_idx; + f.count = cur_count; + f.is_object = cur_is_object; } else { - cold.deep.push_back(f); + cold.deep.push_back(frame{cur_idx, cur_count, cur_is_object}); } } ++depth; @@ -1378,19 +1788,21 @@ indent_done: n.next = static_cast(out - base) - cur_idx; if (--depth != 0) { - frame f{}; if (NLOHMANN_VIEW_LIKELY(depth <= 64)) { - f = cold.shallow[depth - 1]; + const frame& f = cold.shallow[depth - 1]; + cur_idx = f.idx; + cur_count = f.count; + cur_is_object = f.is_object; } else { - f = cold.deep.back(); + const frame f = cold.deep.back(); cold.deep.pop_back(); + cur_idx = f.idx; + cur_count = f.count; + cur_is_object = f.is_object; } - cur_idx = f.idx; - cur_count = f.count; - cur_is_object = f.is_object; } } @@ -1519,12 +1931,13 @@ indent_done: return true; } - /// a string (value or key) at p + /// a string at p: a value (Value) or a key + template NLOHMANN_VIEW_ALWAYS_INLINE bool string() { ++p; // opening quote const unsigned char* const s = p; - p = scan_string_run(p, e); + p = scan_string_run(p, e); if (NLOHMANN_VIEW_LIKELY(p != e && *p == '"')) { emit(value_t::string, 0, 0, static_cast(s - b), static_cast(p - s)); @@ -2377,6 +2790,142 @@ NLOHMANN_JSON_NAMESPACE_END // #include +// #include +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + + + +#include // size_t +#include // uint32_t, uint64_t +#include // memcmp + +// #include +// #include + +// #include + +// #include + + +// Hash indexes of large objects, so that a lookup does not compare thousands +// of keys (as Boost.JSON switches from a linear search to a hash table for +// large objects). An object with document_data::index_min_members members or +// more gets an open-addressing table after parsing; its node stores the +// number of the table (1-based) in `extra`. A slot holds the offset of a key +// node from its object node (0: empty). Of duplicate keys, the first is kept, +// as for the linear search. + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ +namespace view +{ + +/// hash of a key: its bytes, eight at a time, in a fixed byte order +inline std::uint64_t key_hash(const char* s, std::size_t n) noexcept +{ + std::uint64_t h = 0x9E3779B97F4A7C15u * (n + 1); + const auto* p = reinterpret_cast(s); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + while (n >= 8) + { + h = (h ^ read_eight_bytes(p)) * 0xBF58476D1CE4E5B9u; + h ^= h >> 29u; + p += 8; + n -= 8; + } + std::uint64_t w = 0; + for (std::size_t i = 0; i < n; ++i) + { + w |= static_cast(p[i]) << (8u * i); + } + h = (h ^ w) * 0x94D049BB133111EBu; + return h ^ (h >> 31u); +} + +/// build the table of a large object +inline void build_object_index(document_data& d, node* obj) +{ + if (d.indexes.size() >= 0xFFFFu) + { + return; // LCOV_EXCL_LINE (the number must fit `extra`; more large objects are searched linearly) + } + std::size_t cap = 16; + while (cap < 2 * static_cast(obj->len)) + { + cap *= 2; + } + const std::size_t start = d.index_slots.size(); + d.index_slots.resize(start + cap, 0); + std::uint32_t* const slots = d.index_slots.data() + start; + const std::size_t mask = cap - 1; + for (const node* k = document_data::first_child(obj), *end = document_data::child_end(obj); k != end; k = document_data::after(k + 1)) + { + const char* const key = d.str(*k); + const std::uint64_t hash = key_hash(key, k->len); // (a cast of the call would be useless where std::uint64_t is std::size_t) + std::size_t i = static_cast(hash) & mask; + bool duplicate = false; + while (slots[i] != 0) + { + const node* const other = obj + slots[i]; + if (other->len == k->len && (k->len == 0 || std::memcmp(d.str(*other), key, k->len) == 0)) + { + duplicate = true; // keep the first + break; + } + i = (i + 1) & mask; + } + if (!duplicate) + { + slots[i] = static_cast(k - obj); + } + } + d.indexes.push_back(document_data::object_index{start, static_cast(mask)}); + obj->extra = static_cast(d.indexes.size()); +} + +/// build the tables of the large objects the parser noted +inline void build_object_indexes(document_data& d) +{ + for (const std::uint32_t i : d.large_objects) + { + build_object_index(d, d.tape + i); + } +} + +/// the key node of the first member with this key of an indexed object, or +/// nullptr +inline const node* find_indexed(const document_data& d, const node* obj, const char* key, std::size_t n) noexcept +{ + const document_data::object_index& ix = d.indexes[obj->extra - 1u]; + const std::uint32_t* const slots = d.index_slots.data() + ix.start; + const std::uint64_t hash = key_hash(key, n); // (a cast of the call would be useless where std::uint64_t is std::size_t) + std::size_t i = static_cast(hash) & ix.mask; + for (;;) + { + const std::uint32_t s = slots[i]; + if (s == 0) + { + return nullptr; + } + const node* const k = obj + s; + if (k->len == n && (n == 0 || std::memcmp(d.str(*k), key, n) == 0)) + { + return k; + } + i = (i + 1) & ix.mask; + } +} + +} // namespace view +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END + NLOHMANN_JSON_NAMESPACE_BEGIN namespace detail @@ -2446,6 +2995,10 @@ class short_key /// nullptr; most keys are rejected by their length, from the index alone inline const node* find_member(const document_data& d, const node* object, const char* key, std::size_t n) noexcept { + if (NLOHMANN_VIEW_UNLIKELY(object->extra != 0)) + { + return find_indexed(d, object, key, n); // a large object + } const node* const end = document_data::child_end(object); const auto* const k = reinterpret_cast(key); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) if (NLOHMANN_VIEW_LIKELY(n <= 16)) @@ -2805,6 +3358,8 @@ NLOHMANN_JSON_NAMESPACE_END // #include +// #include + // #include // __ _____ _____ _____ // __| | __| | | | JSON for Modern C++ @@ -4502,7 +5057,9 @@ class basic_json_document } return sizeof(document_data) + (m_data->inline_cap * sizeof(detail::view::node)) + (m_data->tape != m_data->inline_tape ? m_data->tape_cap * sizeof(detail::view::node) : 0) - + m_data->arena.capacity() + m_data->owned.capacity(); + + m_data->arena.capacity() + m_data->owned.capacity() + + (m_data->indexes.capacity() * sizeof(document_data::object_index)) + (m_data->index_slots.capacity() * sizeof(std::uint32_t)) + + (m_data->large_objects.capacity() * sizeof(std::uint32_t)); } /// release unused capacity of the index and the decoded strings; like @@ -4572,6 +5129,9 @@ class basic_json_document d.size = size; d.tape_size = 0; d.arena.clear(); + d.indexes.clear(); + d.index_slots.clear(); + d.large_objects.clear(); d.discarded = true; detail::view::parse_failure failure; bool ok = false; @@ -4587,6 +5147,7 @@ class basic_json_document { d.base[0] = d.src; d.base[1] = d.arena.data(); + detail::view::build_object_indexes(d); d.discarded = false; return; } @@ -4754,6 +5315,13 @@ class tuple_element> // NOLINT(cert #undef NLOHMANN_VIEW_THROW #undef NLOHMANN_VIEW_LITTLE_ENDIAN #undef NLOHMANN_VIEW_REPEAT16 +#undef NLOHMANN_VIEW_NEON +#undef NLOHMANN_VIEW_SSE2 +#undef NLOHMANN_VIEW_SSSE3 +#undef NLOHMANN_VIEW_SSSE3_DISPATCH +#undef NLOHMANN_VIEW_SSSE3_TARGET +#undef NLOHMANN_VIEW_VECTOR +#undef NLOHMANN_VIEW_VECTOR_UTF8 #endif // INCLUDE_NLOHMANN_JSON_VIEW_HPP_ diff --git a/tests/CMakeLists.txt b/tests/CMakeLists.txt index 8735934de..64717a246 100644 --- a/tests/CMakeLists.txt +++ b/tests/CMakeLists.txt @@ -363,6 +363,25 @@ json_test_add_test_for(src/unit-diagnostic-positions.cpp MAIN test_main CXX_STANDARDS ${test_cxx_standards} ${test_force} ) +# the json_view parser again with the portable string scanning instead of +# NEON/SSE2, and on x86-64 with the SSSE3 UTF-8 check (JSON_VIEW_USE_SSSE3) +json_test_set_test_options(test-json_view_builder_portable + COMPILE_DEFINITIONS JSON_VIEW_NO_SIMD +) +json_test_add_test_for(src/unit-json_view_builder.cpp + NAME test-json_view_builder_portable + MAIN test_main CXX_STANDARDS ${test_cxx_standards} ${test_force} +) +if(CMAKE_SYSTEM_PROCESSOR MATCHES "^(x86_64|AMD64|amd64)$" AND NOT MSVC) + json_test_set_test_options(test-json_view_builder_ssse3 + COMPILE_DEFINITIONS JSON_VIEW_USE_SSSE3 + COMPILE_OPTIONS -mssse3 + ) + json_test_add_test_for(src/unit-json_view_builder.cpp + NAME test-json_view_builder_ssse3 + MAIN test_main CXX_STANDARDS ${test_cxx_standards} ${test_force} + ) +endif() endif() # *DO NOT* use json_test_set_test_options() below this line diff --git a/tests/src/unit-json_view.cpp b/tests/src/unit-json_view.cpp index cea863c66..9494fb3e8 100644 --- a/tests/src/unit-json_view.cpp +++ b/tests/src/unit-json_view.cpp @@ -1277,3 +1277,60 @@ TEST_CASE("json_view comparison") CHECK(a.root() != json_document::parse(other).root()); } } + +TEST_CASE("json_view large objects") +{ + // objects with 128 members or more are looked up with a hash index + for (const std::size_t members : + { + 127u, 128u, 129u, 10000u + }) + { + CAPTURE(members) + std::string text = "{"; + for (std::size_t i = 0; i < members; ++i) + { + text += (i != 0 ? ",\"" : "\"") + std::string(i % 23, 'k') + std::to_string(i) + (i % 7 == 0 ? "\\n" : "") + "\":" + std::to_string(i); + } + text += R"(,"":"empty key","k1":"a duplicate of an earlier key"})"; + const json_document d = json_document::parse(text); + const json_view v = d.root(); + const json j = json::parse(text); + for (std::size_t i = 0; i < members; ++i) + { + const std::string key = std::string(i % 23, 'k') + std::to_string(i) + (i % 7 == 0 ? "\n" : ""); + CHECK(v[key].get() == i); + CHECK(v.contains(key)); + CHECK(v.find(key).key() == key); + CHECK(v.at(key).get() == i); + CHECK(!v.contains(key + "x")); + } + CHECK(v[""].get_string() == "empty key"); + CHECK(v["k1"].get() == 1); // the first of duplicate keys, as for small objects + CHECK(!v.contains("missing")); + CHECK_THROWS_WITH_AS(v.at("missing"), "[json.exception.out_of_range.403] key 'missing' not found", json::out_of_range&); + CHECK(v == j); + CHECK(v.materialize() == j); + } + + SECTION("nested, reused, and in arrays") + { + std::string inner = "{"; + for (int i = 0; i < 300; ++i) + { + inner += (i != 0 ? ",\"m" : "\"m") + std::to_string(i) + "\":" + std::to_string(i); + } + inner += '}'; + const std::string text = "[" + inner + ",{\"x\":" + inner + "}," + inner + "]"; + json_document d = json_document::parse(text); + CHECK(d.root()[0]["m299"].get() == 299); + CHECK(d.root()[1]["x"]["m150"].get() == 150); + CHECK(d.root()[2]["m0"].get() == 0); + const std::size_t with_index = d.memory_usage(); + d.read(std::string("{\"small\": 1}")); + CHECK(d.root()["small"].get() == 1); + d.read(text); + CHECK(d.root()[2]["m7"].get() == 7); + CHECK(d.memory_usage() >= with_index / 2); + } +} diff --git a/tests/src/unit-json_view_builder.cpp b/tests/src/unit-json_view_builder.cpp index 9d3ea943b..36f3d7c17 100644 --- a/tests/src/unit-json_view_builder.cpp +++ b/tests/src/unit-json_view_builder.cpp @@ -17,6 +17,7 @@ #endif using nlohmann::json; +#include #include #include #include @@ -142,6 +143,32 @@ void check_same(const std::string& text) } } +// a string value and a key must be accepted or rejected as json::parse does, +// and give its value (cheaper than check_same: the options do not matter) +void check_string(const std::string& content) +{ + for (const std::string& text : + { + "[\"" + content + "\"]", "{\"" + content + "\":1}" + }) + { + const bool accepted = json::accept(text); + for (const bool sentinel : + { + true, false + }) + { + const built b = build(text, false, false, sentinel); + if (b.ok != accepted || (accepted && value_of(b) != json::parse(text))) + { + CAPTURE(text) + CHECK(b.ok == accepted); + CHECK((b.ok && accepted ? value_of(b) == json::parse(text) : true)); + } + } + } +} + // a small deterministic generator of documents struct generator { @@ -401,3 +428,81 @@ TEST_CASE("json_view builder") } } } + +TEST_CASE("json_view builder: strings across vector blocks") +{ + // Strings are scanned 8 or 16 bytes at a time (NEON, SSE2, or SWAR) and + // non-ASCII text with the vector UTF-8 check (NEON, SSSE3) or one sequence + // at a time. Sequences are placed so that they start at every offset + // around the block boundaries of keys (16, 32) and values (8, 24), with + // text of several lengths after them. + const std::array prefixes = {{0, 6, 7, 8, 13, 14, 15, 16, 21, 22, 23, 24, 29, 30, 31, 32}}; + const std::array suffixes = {{0, 3, 17}}; + const auto around = [&](const std::string & seq, std::size_t prefix, std::size_t suffix) + { + return std::string(prefix, 'a') + seq + std::string(suffix, 'b'); + }; + + SECTION("every two-byte sequence") + { + for (unsigned lead = 0x80; lead <= 0xFF; ++lead) + { + for (unsigned second = 0; second <= 0xFF; ++second) + { + if (second == '"' || second == '\\') + { + continue; + } + const std::string seq = {static_cast(lead), static_cast(second)}; + check_string(around(seq, prefixes[(lead + second) % 16], suffixes[second % 3])); + } + } + } + + SECTION("three- and four-byte sequences") + { + const std::array conts = {{0x7F, 0x80, 0x8F, 0x90, 0x9F, 0xA0, 0xBF, 0xC0}}; + for (unsigned lead = 0xE0; lead <= 0xF7; ++lead) + { + for (const unsigned b2 : conts) + { + for (const unsigned b3 : conts) + { + for (const std::size_t prefix : prefixes) + { + std::string seq = {static_cast(lead), static_cast(b2), static_cast(b3)}; + if (lead >= 0xF0) + { + seq += static_cast(prefix % 2 == 0 ? 0x80 : 0xBF); + } + check_string(around(seq, prefix, suffixes[prefix % 3])); + // cut short before the end of the string + check_string(around(seq.substr(0, seq.size() - 1), prefix, suffixes[prefix % 3])); + } + } + } + } + } + + SECTION("long runs of text with one damaged byte") + { + const std::array chars = {{"a", "\xc3\xa9", "\xe3\x81\x82", "\xf0\x9f\x98\x80", "\xed\x9f\xbf", "\xef\xbf\xbf", "\xf4\x8f\xbf\xbf"}}; + const std::array damage = {{'\x80', '\xbf', '\xc0', '\xc1', '\xe0', '\xed', '\xf5', '\xff', '\x1f', '"', '\\'}}; + std::mt19937 rng(5295); // NOLINT(cert-msc32-c,cert-msc51-cpp,bugprone-random-generator-seed): reproducible + for (int i = 0; i < 4000; ++i) + { + std::string text; + const auto n = rng() % 60; + for (unsigned k = 0; k < n; ++k) + { + text += chars[rng() % chars.size()]; + } + check_string(text); + if (!text.empty()) + { + text[rng() % text.size()] = damage[rng() % damage.size()]; + check_string(text); + } + } + } +}