From fdcc569eee979cb8237c5a2c61e290c527b2d487 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Wed, 30 Sep 2026 20:13:31 +0200 Subject: [PATCH 1/7] Use only documented StringType members in json_pointer (#5692) contains(const json_pointer&) and operator/=(std::size_t) (and hence operator/(std::size_t)) used string_t operations that the StringType template parameter documentation explicitly does not require: comparing string_t with a const char* literal, c_str(), and constructibility from std::string. This made both functions fail to compile for a conforming custom StringType, even though the documentation's own reference StringType satisfies the requirements. Fix contains() to compare individual chars ('0'..'9') instead of comparing string_t with const char* literals, and to call data() (documented to be null-terminated) instead of c_str(). Fix operator/=(std::size_t) to build the array-index token via the existing detail::to_string helper (ADL int_to_string() or assignment from std::to_string()) instead of via std::to_string() directly, matching how diff(), items(), and std::hash already convert a std::size_t to a StringType. Add regression tests to tests/src/unit-alt-string.cpp: contains() for present/missing keys and indices, "-", a leading zero, and a non-numeric token on an array, plus json_pointer::operator/(std::size_t). Fixes #5666. Signed-off-by: Niels Lohmann --- .../features/types/template_parameters.md | 1 + include/nlohmann/detail/json_pointer.hpp | 7 ++-- single_include/nlohmann/json.hpp | 8 +++-- tests/src/unit-alt-string.cpp | 35 +++++++++++++++++++ 4 files changed, 45 insertions(+), 6 deletions(-) diff --git a/docs/mkdocs/docs/features/types/template_parameters.md b/docs/mkdocs/docs/features/types/template_parameters.md index 8ed23e156..5b528fd79 100644 --- a/docs/mkdocs/docs/features/types/template_parameters.md +++ b/docs/mkdocs/docs/features/types/template_parameters.md @@ -389,6 +389,7 @@ using array_t = ArrayType>; | Functionality | Additional requirement | |-----------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------| | [`diff`](../../api/basic_json/diff.md), [`items`](../../api/basic_json/items.md), [`std::hash`](../../api/basic_json/std_hash.md) | conversion of a `#!cpp std::size_t` to `StringType`: either assignability from the result of `#!cpp std::to_string`, or an ADL overload `#!cpp void int_to_string(StringType&, std::size_t)` | +| [`operator/(std::size_t)`](../../api/json_pointer/operator_slash.md) | the same conversion of a `#!cpp std::size_t` to `StringType` as `diff`, `items`, and `std::hash` above | | [`std::hash`](../../api/basic_json/std_hash.md) | additionally a specialization of `#!cpp std::hash` | | [`to_bson`](../../api/basic_json/to_bson.md) | `find(value_type)` and `npos` | | [`parse`](../../api/basic_json/parse.md) from a `string_t` | the input adapters must accept it; otherwise pass a character range | diff --git a/include/nlohmann/detail/json_pointer.hpp b/include/nlohmann/detail/json_pointer.hpp index 0b9f9651a..78074988e 100644 --- a/include/nlohmann/detail/json_pointer.hpp +++ b/include/nlohmann/detail/json_pointer.hpp @@ -26,6 +26,7 @@ #include #include #include +#include #include NLOHMANN_JSON_NAMESPACE_BEGIN @@ -116,7 +117,7 @@ class json_pointer /// @sa https://json.nlohmann.me/api/json_pointer/operator_slasheq/ json_pointer& operator/=(std::size_t array_idx) { - return *this /= std::to_string(array_idx); + return *this /= detail::to_string(array_idx); } /// @brief create a new JSON pointer by appending the right JSON pointer at the end of the left JSON pointer @@ -752,7 +753,7 @@ class json_pointer // would throw out_of_range.404 -- contains() must not throw (see #5395) return false; } - if (JSON_HEDLEY_UNLIKELY(reference_token.size() == 1 && !("0" <= reference_token && reference_token <= "9"))) + if (JSON_HEDLEY_UNLIKELY(reference_token.size() == 1 && !('0' <= reference_token[0] && reference_token[0] <= '9'))) { // invalid char return false; @@ -780,7 +781,7 @@ class json_pointer // not throw (see #5395), so such a reference token is treated as "not found" errno = 0; // strtoull() does not reset errno on success char* p_end = nullptr; // NOLINT(misc-const-correctness) - const unsigned long long magnitude = std::strtoull(reference_token.c_str(), &p_end, 10); // NOLINT(runtime/int) + const unsigned long long magnitude = std::strtoull(reference_token.data(), &p_end, 10); // NOLINT(runtime/int) if (JSON_HEDLEY_UNLIKELY(errno == ERANGE // the value exceeds ULLONG_MAX || magnitude >= static_cast((std::numeric_limits::max)()))) // NOLINT(runtime/int) { diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index cf94c53a9..88c29a4f1 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -19636,6 +19636,8 @@ NLOHMANN_JSON_NAMESPACE_END // #include +// #include + // #include @@ -19727,7 +19729,7 @@ class json_pointer /// @sa https://json.nlohmann.me/api/json_pointer/operator_slasheq/ json_pointer& operator/=(std::size_t array_idx) { - return *this /= std::to_string(array_idx); + return *this /= detail::to_string(array_idx); } /// @brief create a new JSON pointer by appending the right JSON pointer at the end of the left JSON pointer @@ -20363,7 +20365,7 @@ class json_pointer // would throw out_of_range.404 -- contains() must not throw (see #5395) return false; } - if (JSON_HEDLEY_UNLIKELY(reference_token.size() == 1 && !("0" <= reference_token && reference_token <= "9"))) + if (JSON_HEDLEY_UNLIKELY(reference_token.size() == 1 && !('0' <= reference_token[0] && reference_token[0] <= '9'))) { // invalid char return false; @@ -20391,7 +20393,7 @@ class json_pointer // not throw (see #5395), so such a reference token is treated as "not found" errno = 0; // strtoull() does not reset errno on success char* p_end = nullptr; // NOLINT(misc-const-correctness) - const unsigned long long magnitude = std::strtoull(reference_token.c_str(), &p_end, 10); // NOLINT(runtime/int) + const unsigned long long magnitude = std::strtoull(reference_token.data(), &p_end, 10); // NOLINT(runtime/int) if (JSON_HEDLEY_UNLIKELY(errno == ERANGE // the value exceeds ULLONG_MAX || magnitude >= static_cast((std::numeric_limits::max)()))) // NOLINT(runtime/int) { diff --git a/tests/src/unit-alt-string.cpp b/tests/src/unit-alt-string.cpp index f95d2213c..cac2d183d 100644 --- a/tests/src/unit-alt-string.cpp +++ b/tests/src/unit-alt-string.cpp @@ -352,6 +352,41 @@ TEST_CASE("alternative string type") CHECK(j2.flatten().unflatten() == j2); } + SECTION("contains(json_pointer)") + { + // contains(json_pointer) must compile and work with a string_t that has + // no c_str() and no comparison with const char* (see #5666) + auto j = alt_json::parse(R"({"foo": ["bar", "baz"]})"); + + // present: object key and array indices + CHECK(j.contains(alt_json::json_pointer("/foo"))); + CHECK(j.contains(alt_json::json_pointer("/foo/0"))); + CHECK(j.contains(alt_json::json_pointer("/foo/1"))); + + // missing: absent object key and out-of-range array index + CHECK_FALSE(j.contains(alt_json::json_pointer("/bar"))); + CHECK_FALSE(j.contains(alt_json::json_pointer("/foo/2"))); + + // "-" always fails the range check + CHECK_FALSE(j.contains(alt_json::json_pointer("/foo/-"))); + + // an array index must not have a leading zero + CHECK_FALSE(j.contains(alt_json::json_pointer("/foo/01"))); + + // a reference token that is not a number + CHECK_FALSE(j.contains(alt_json::json_pointer("/foo/bar"))); + } + + SECTION("operator/(std::size_t)") + { + // json_pointer::operator/=(std::size_t) must compile without string_t + // being constructible from std::string (see #5666) + auto j = alt_json::parse(R"({"foo": ["bar", "baz"]})"); + + CHECK(j.at(alt_json::json_pointer("/foo") / std::size_t(0)) == j["foo"][0]); + CHECK(j.at(alt_json::json_pointer("/foo") / std::size_t(1)) == j["foo"][1]); + } + SECTION("patch") { alt_json const patch1 = alt_json::parse(R"([{ "op": "add", "path": "/a/b", "value": [ "foo", "bar" ] }])"); From 6a073dbae4a2cce5ade27546a8ca365b0fef0786 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Wed, 30 Sep 2026 20:13:34 +0200 Subject: [PATCH 2/7] Give operator>> a strong exception-safety guarantee (#5695) operator>> parsed directly into its basic_json& target, so a parse error left the target holding whatever was parsed before the error instead of its previous value. With JSON_DIAGNOSTICS=1, that partial value also violated the class invariant, because the parent pointers of an array or object's elements are only set when the container is closed, which a failed parse never reaches; copying such a value then aborted in assert_invariant(). Fix it the way basic_json::parse() already handles this: parse into a temporary and move it into the target only once parsing succeeds, so the target is left unchanged if an exception is thrown. Fixes #5652. Signed-off-by: Niels Lohmann --- docs/mkdocs/docs/api/operator_gtgt.md | 6 ++++++ include/nlohmann/json.hpp | 5 ++++- single_include/nlohmann/json.hpp | 5 ++++- tests/src/unit-diagnostics.cpp | 16 ++++++++++++++++ 4 files changed, 30 insertions(+), 2 deletions(-) diff --git a/docs/mkdocs/docs/api/operator_gtgt.md b/docs/mkdocs/docs/api/operator_gtgt.md index b68889af9..35b8c1016 100644 --- a/docs/mkdocs/docs/api/operator_gtgt.md +++ b/docs/mkdocs/docs/api/operator_gtgt.md @@ -18,6 +18,10 @@ Deserializes an input stream to a JSON value. the stream `i` +## Exception safety + +Strong guarantee: if an exception is thrown, there are no changes in `j`. + ## Exceptions - Throws [`parse_error.101`](../home/exceptions.md#jsonexceptionparse_error101) in case of an unexpected token, or if @@ -125,3 +129,5 @@ being read. the stream; planned to become the default in version 4.0.0. - Fixed a null pointer dereference for an `std::istream` without a stream buffer (now throws `parse_error.101`), and a crash (`std::terminate`) when `i` has `eofbit` in its exception mask, in version 3.13.0. +- Changed to the strong exception safety guarantee in version 3.13.0: `j` is no longer left with a partially parsed + value if parsing throws. diff --git a/include/nlohmann/json.hpp b/include/nlohmann/json.hpp index cb08777bc..78c4e4f62 100644 --- a/include/nlohmann/json.hpp +++ b/include/nlohmann/json.hpp @@ -5092,7 +5092,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @sa https://json.nlohmann.me/api/basic_json/operator_gtgt/ friend std::istream& operator>>(std::istream& i, basic_json& j) { - parser(detail::input_adapter(i)).parse(false, j); + // parse into a temporary so that j is left unchanged if parsing fails + basic_json result; + parser(detail::input_adapter(i)).parse(false, result); + j = std::move(result); return i; } #endif // JSON_NO_IO diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 88c29a4f1..5ef598150 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -32019,7 +32019,10 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @sa https://json.nlohmann.me/api/basic_json/operator_gtgt/ friend std::istream& operator>>(std::istream& i, basic_json& j) { - parser(detail::input_adapter(i)).parse(false, j); + // parse into a temporary so that j is left unchanged if parsing fails + basic_json result; + parser(detail::input_adapter(i)).parse(false, result); + j = std::move(result); return i; } #endif // JSON_NO_IO diff --git a/tests/src/unit-diagnostics.cpp b/tests/src/unit-diagnostics.cpp index 46f41f252..b3778e802 100644 --- a/tests/src/unit-diagnostics.cpp +++ b/tests/src/unit-diagnostics.cpp @@ -19,6 +19,7 @@ using nlohmann::json; #include #include +#include TEST_CASE("Better diagnostics") { @@ -492,6 +493,21 @@ TEST_CASE("Regression tests for extended diagnostics") CHECK(copy == j); } } + + SECTION("Regression test for issue #5652 - operator>> leaves a partial value in its target on a parse error") + { + json j = "old value"; + std::istringstream is("[1, x"); + CHECK_THROWS_WITH_AS(is >> j, "[json.exception.parse_error.101] parse error at line 1, column 5: syntax error while parsing value - invalid literal; last read: '1, x'", json::parse_error); + + // j must be left unchanged, as json::parse() guarantees for its result + CHECK(j == "old value"); + + // copying j must not trigger assert_invariant(): a failed parse must + // not leave array/object elements without a parent pointer + json const copy = j; // NOLINT(performance-unnecessary-copy-initialization) + CHECK(copy == j); + } } TEST_CASE("Better diagnostics past the descent bound of update() and merge_patch()") From 3b6ae43c539397fe664929e881c6c2cfd0a331b7 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Wed, 30 Sep 2026 20:53:56 +0200 Subject: [PATCH 3/7] Add JSON_DISABLE_TUPLE_REFERENCE_CONVERSION to fix std::tuple conversions (#5598) * Add JSON_DISABLE_TUPLE_REFERENCE_CONVERSION to fix std::tuple conversions basic_json can be constructed from std::tuple, which it turns into a one-element array. Because of this, std::tuple picks its converting constructor that converts the whole source tuple instead of the element-wise one. As a result, std::tuple built from std::forward_as_tuple(j) binds to a temporary (a compile error with libc++, a dangling reference with other standard libraries), and std::tuple built the same way holds [j] instead of a copy of j. The new opt-in macro JSON_DISABLE_TUPLE_REFERENCE_CONVERSION (CMake option JSON_DisableTupleReferenceConversion) removes the conversion from a one-element tuple holding a reference to the same basic_json type, so std::tuple converts element-wise. It is off by default, so existing behavior is unchanged. It does not change any function body and therefore is not part of the ABI tag. Fixes #2226 Signed-off-by: Niels Lohmann * Convert one-element tuples to arrays on every compiler to_json for std::tuple assigns a braced list, j = { std::get(t)... }. With a single element that is itself a basic_json, Apple clang 15 and 16 treat j = {x} as a copy of x, so std::tuple{true} became true instead of [true]. The macOS jobs (Xcode 15.1, 16.1) failed the new checks in unit-disable-tuple-reference-conversion and unit-regression2. The one-element overload that already handles JSON_BRACE_INIT_COPY_SEMANTICS builds the array (or object, for a [string, value] element) explicitly, the same way the initializer-list constructor does. Use it unconditionally. The output is unchanged on compilers that already wrapped the element. Signed-off-by: Niels Lohmann * Skip json reference tuple tests on clang < 4 and GCC < 5 ci_test_compilers_gcc_old (4.8) and ci_test_compilers_clang (3.4) could not compile the new tuple tests. Creating a std::tuple of basic_json references, e.g. std::forward_as_tuple(j), makes these compilers instantiate basic_json's conversion operator for libstdc++'s internal tuple bases, which fails hard. This happens with and without JSON_DISABLE_TUPLE_REFERENCE_CONVERSION, so it is a limitation of these compilers, not of the new option. Tested with the CI images: clang 3.4 to 3.9 and GCC 4.8 and 4.9 fail, clang 4, 5, and 6 and GCC 5 and 6 compile all cases. Skip only the checks that create such tuples; the is_constructible checks still run. Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- .github/workflows/ubuntu.yml | 2 +- CMakeLists.txt | 6 + cmake/ci.cmake | 14 +++ docs/docset/docSet.sql | 1 + docs/mkdocs/docs/api/basic_json/basic_json.md | 2 + docs/mkdocs/docs/api/macros/index.md | 1 + ...json_disable_tuple_reference_conversion.md | 114 ++++++++++++++++++ docs/mkdocs/docs/features/macros.md | 7 ++ docs/mkdocs/docs/integration/cmake.md | 6 + docs/mkdocs/mkdocs.yml | 1 + .../nlohmann/detail/conversions/to_json.hpp | 13 +- include/nlohmann/detail/macro_scope.hpp | 4 + include/nlohmann/detail/macro_unscope.hpp | 1 + include/nlohmann/detail/meta/type_traits.hpp | 12 ++ include/nlohmann/json.hpp | 7 +- single_include/nlohmann/json.hpp | 37 ++++-- ...nit-disable-tuple-reference-conversion.cpp | 93 ++++++++++++++ tests/src/unit-regression2.cpp | 28 +++++ 18 files changed, 334 insertions(+), 15 deletions(-) create mode 100644 docs/mkdocs/docs/api/macros/json_disable_tuple_reference_conversion.md create mode 100644 tests/src/unit-disable-tuple-reference-conversion.cpp diff --git a/.github/workflows/ubuntu.yml b/.github/workflows/ubuntu.yml index d8c27c621..d8f15e3dd 100644 --- a/.github/workflows/ubuntu.yml +++ b/.github/workflows/ubuntu.yml @@ -100,7 +100,7 @@ jobs: container: ubuntu:focal strategy: matrix: - target: [ci_cmake_flags, ci_test_diagnostics, ci_test_diagnostic_positions, ci_test_noexceptions, ci_test_noimplicitconversions, ci_test_legacycomparison, ci_test_noglobaludls, ci_test_disableenumserialization, ci_test_skiplibraryversioncheck, ci_test_simdutf, ci_test_strict_nul_handling, ci_test_no_thread_local] + target: [ci_cmake_flags, ci_test_diagnostics, ci_test_diagnostic_positions, ci_test_noexceptions, ci_test_noimplicitconversions, ci_test_legacycomparison, ci_test_noglobaludls, ci_test_disableenumserialization, ci_test_disabletuplereferenceconversion, ci_test_skiplibraryversioncheck, ci_test_simdutf, ci_test_strict_nul_handling, ci_test_no_thread_local] steps: - name: Install build-essential run: apt-get update ; apt-get install -y build-essential unzip wget git libssl-dev diff --git a/CMakeLists.txt b/CMakeLists.txt index 4c43c23ff..f0a6771dc 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -55,6 +55,7 @@ option(JSON_Diagnostic_Positions "Enable diagnostic positions." OFF) option(JSON_GlobalUDLs "Place user-defined string literals in the global namespace." ON) option(JSON_ImplicitConversions "Enable implicit conversions." ON) option(JSON_DisableEnumSerialization "Disable default integer enum serialization." OFF) +option(JSON_DisableTupleReferenceConversion "Disable conversion from a one-element tuple of a JSON reference." OFF) option(JSON_LegacyDiscardedValueComparison "Enable legacy discarded value comparison." OFF) option(JSON_Install "Install CMake targets during install step." ${MAIN_PROJECT}) option(JSON_MultipleHeaders "Use non-amalgamated version of the library." ON) @@ -101,6 +102,10 @@ if (JSON_DisableEnumSerialization) message(STATUS "Enum integer serialization is disabled (JSON_DISABLE_ENUM_SERIALIZATION=1)") endif() +if (JSON_DisableTupleReferenceConversion) + message(STATUS "Tuple reference conversion is disabled (JSON_DISABLE_TUPLE_REFERENCE_CONVERSION=1)") +endif() + if (JSON_LegacyDiscardedValueComparison) message(STATUS "Legacy discarded value comparison enabled (JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON=1)") endif() @@ -143,6 +148,7 @@ target_compile_definitions( $<$>:JSON_USE_GLOBAL_UDLS=0> $<$>:JSON_USE_IMPLICIT_CONVERSIONS=0> $<$:JSON_DISABLE_ENUM_SERIALIZATION=1> + $<$:JSON_DISABLE_TUPLE_REFERENCE_CONVERSION=1> $<$:JSON_DIAGNOSTICS=1> $<$:JSON_DIAGNOSTIC_POSITIONS=1> $<$:JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON=1> diff --git a/cmake/ci.cmake b/cmake/ci.cmake index a001a05ae..27187f235 100644 --- a/cmake/ci.cmake +++ b/cmake/ci.cmake @@ -276,6 +276,20 @@ add_custom_target(ci_test_disableenumserialization COMMENT "Compile and test with enum serialization disabled" ) +############################################################################### +# Disable conversion from a one-element tuple of a JSON reference. +############################################################################### + +add_custom_target(ci_test_disabletuplereferenceconversion + COMMAND ${CMAKE_COMMAND} + -DCMAKE_BUILD_TYPE=Debug -GNinja + -DJSON_BuildTests=ON -DJSON_FastTests=ON -DJSON_DisableTupleReferenceConversion=ON + -S${PROJECT_SOURCE_DIR} -B${PROJECT_BINARY_DIR}/build_disabletuplereferenceconversion + COMMAND ${CMAKE_COMMAND} --build ${PROJECT_BINARY_DIR}/build_disabletuplereferenceconversion + COMMAND cd ${PROJECT_BINARY_DIR}/build_disabletuplereferenceconversion && ${CMAKE_CTEST_COMMAND} --parallel ${N} --output-on-failure + COMMENT "Compile and test with tuple reference conversion disabled" +) + ############################################################################### # Skip the multiple-inclusion library version check. ############################################################################### diff --git a/docs/docset/docSet.sql b/docs/docset/docSet.sql index 3a97e405f..6ec90dcfc 100644 --- a/docs/docset/docSet.sql +++ b/docs/docset/docSet.sql @@ -210,6 +210,7 @@ INSERT INTO searchIndex(name, type, path) VALUES ('JSON_CATCH_USER', 'Macro', 'a INSERT INTO searchIndex(name, type, path) VALUES ('JSON_DIAGNOSTICS', 'Macro', 'api/macros/json_diagnostics/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('JSON_DIAGNOSTIC_POSITIONS', 'Macro', 'api/macros/json_diagnostic_positions/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('JSON_DISABLE_ENUM_SERIALIZATION', 'Macro', 'api/macros/json_disable_enum_serialization/index.html'); +INSERT INTO searchIndex(name, type, path) VALUES ('JSON_DISABLE_TUPLE_REFERENCE_CONVERSION', 'Macro', 'api/macros/json_disable_tuple_reference_conversion/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('JSON_HAS_CPP_11', 'Macro', 'api/macros/json_has_cpp_11/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('JSON_HAS_CPP_14', 'Macro', 'api/macros/json_has_cpp_11/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('JSON_HAS_CPP_17', 'Macro', 'api/macros/json_has_cpp_11/index.html'); diff --git a/docs/mkdocs/docs/api/basic_json/basic_json.md b/docs/mkdocs/docs/api/basic_json/basic_json.md index 3a7a6eb8b..962f3e164 100644 --- a/docs/mkdocs/docs/api/basic_json/basic_json.md +++ b/docs/mkdocs/docs/api/basic_json/basic_json.md @@ -159,6 +159,8 @@ basic_json(basic_json&& other) noexcept; - `CompatibleType` is not `basic_json` (to avoid hijacking copy/move constructors), - `CompatibleType` is not a different `basic_json` type (i.e. with different template arguments) - `CompatibleType` is not a `basic_json` nested type (e.g., `json_pointer`, `iterator`, etc.) + - if [`JSON_DISABLE_TUPLE_REFERENCE_CONVERSION`](../macros/json_disable_tuple_reference_conversion.md) is defined + to `1`: `CompatibleType` is not a one-element `std::tuple` holding a reference to `basic_json` - `json_serializer` (with `U = uncvref_t`) has a `to_json(basic_json_t&, CompatibleType&&)` method diff --git a/docs/mkdocs/docs/api/macros/index.md b/docs/mkdocs/docs/api/macros/index.md index 9d2636eb2..e917e0ae2 100644 --- a/docs/mkdocs/docs/api/macros/index.md +++ b/docs/mkdocs/docs/api/macros/index.md @@ -53,6 +53,7 @@ header. See also the [macro overview page](../../features/macros.md). - [**JSON_BRACE_INIT_COPY_SEMANTICS**](json_brace_init_copy_semantics.md) - opt in to copy/move semantics for single-element brace initialization - [**JSON_DISABLE_ENUM_SERIALIZATION**](json_disable_enum_serialization.md) - switch off default serialization/deserialization functions for enums +- [**JSON_DISABLE_TUPLE_REFERENCE_CONVERSION**](json_disable_tuple_reference_conversion.md) - switch off conversion from a one-element tuple of a JSON reference - [**JSON_USE_IMPLICIT_CONVERSIONS**](json_use_implicit_conversions.md) - control implicit conversions ## Comparison behavior diff --git a/docs/mkdocs/docs/api/macros/json_disable_tuple_reference_conversion.md b/docs/mkdocs/docs/api/macros/json_disable_tuple_reference_conversion.md new file mode 100644 index 000000000..c759fd65c --- /dev/null +++ b/docs/mkdocs/docs/api/macros/json_disable_tuple_reference_conversion.md @@ -0,0 +1,114 @@ +# JSON_DISABLE_TUPLE_REFERENCE_CONVERSION + +```cpp +#define JSON_DISABLE_TUPLE_REFERENCE_CONVERSION /* value */ +``` + +When defined to `1`, a `basic_json` value can no longer be constructed from a one-element `std::tuple` whose element is +a reference to that `basic_json` type, such as `std::tuple`, `std::tuple`, or `std::tuple`. +These are the tuples created by `std::forward_as_tuple(j)`. + +## Default definition + +The default value is `0` (disabled — existing behavior is preserved). + +```cpp +#define JSON_DISABLE_TUPLE_REFERENCE_CONVERSION 0 +``` + +## Notes + +!!! note "Background" + + By default, `basic_json` can be constructed from any `std::tuple` whose elements can be converted to JSON; the result + is an array. This includes `std::tuple`, which becomes a one-element array. + + `std::tuple` only converts another tuple element by element if its element type cannot be constructed from the whole + source tuple. Because `json` *can* be constructed from `std::tuple`, `std::tuple` instead converts the whole + tuple into a single `json` value. This has two surprising effects: + + ```cpp + json j = true; + + // rejected by some standard libraries (e.g., libc++); with others, the + // reference binds to a temporary that is destroyed right away + std::tuple t1(std::forward_as_tuple(j)); + + // compiles, but std::get<0>(t2) is [true], not true + std::tuple t2(std::forward_as_tuple(j)); + ``` + + Enabling this macro removes the conversion, so both tuples are converted element by element: `std::get<0>(t1)` + refers to `j`, and `std::get<0>(t2)` is a copy of `j` (see [#2226](https://github.com/nlohmann/json/issues/2226)). + +!!! warning "Opt-in only" + + This macro must be defined **before** including ``. Defining it after the include has no effect. + +!!! note "Affected conversions" + + Only one-element tuples holding a reference to the **same** `basic_json` type are affected. Constructing a JSON value + from them no longer compiles: + + ```cpp + json j = true; + json a = std::forward_as_tuple(j); // error with the macro enabled + json b = json::array({j}); // use this instead: [true] + ``` + + Tuples holding a JSON value (`std::make_tuple(j)`), tuples with more than one element, and tuples holding references + to other types (including other `basic_json` specializations) are converted to arrays as before. + +!!! hint "CMake option" + + This behavior can also be controlled with the CMake option + [`JSON_DisableTupleReferenceConversion`](../../integration/cmake.md#json_disabletuplereferenceconversion) + (`OFF` by default) which defines `JSON_DISABLE_TUPLE_REFERENCE_CONVERSION` accordingly. + +## Examples + +??? example "Default behavior (macro not defined)" + + ```cpp + #include + + using json = nlohmann::json; + + int main() + { + json j = true; + + std::tuple t(std::forward_as_tuple(j)); + // std::get<0>(t) is [true] -- the whole tuple was converted + } + ``` + +??? example "Conversion disabled (macro defined to 1)" + + ```cpp + #define JSON_DISABLE_TUPLE_REFERENCE_CONVERSION 1 + #include + + using json = nlohmann::json; + + int main() + { + json j = true; + + std::tuple t(std::forward_as_tuple(j)); + // std::get<0>(t) is true -- a copy of j + + std::tuple r(std::forward_as_tuple(j)); + // std::get<0>(r) refers to j + } + ``` + +## See also + +- [**basic_json(CompatibleType&&)**](../basic_json/basic_json.md) - the affected constructor +- [:simple-cmake: JSON_DisableTupleReferenceConversion](../../integration/cmake.md#json_disabletuplereferenceconversion) - + CMake option to control the macro + +## Version history + +- Added in version 3.13.0. diff --git a/docs/mkdocs/docs/features/macros.md b/docs/mkdocs/docs/features/macros.md index e4a628c3e..a4479712f 100644 --- a/docs/mkdocs/docs/features/macros.md +++ b/docs/mkdocs/docs/features/macros.md @@ -83,6 +83,13 @@ When defined, default parse and serialize functions for enums are excluded and h See [full documentation of `JSON_DISABLE_ENUM_SERIALIZATION`](../api/macros/json_disable_enum_serialization.md). +## `JSON_DISABLE_TUPLE_REFERENCE_CONVERSION` + +When defined to `1`, a JSON value can no longer be created from a one-element `std::tuple` holding a reference to a JSON +value, such as the result of `std::forward_as_tuple(j)`. This lets `std::tuple` convert such tuples element-wise. + +See [full documentation of `JSON_DISABLE_TUPLE_REFERENCE_CONVERSION`](../api/macros/json_disable_tuple_reference_conversion.md). + ## `JSON_NO_AUTOMATIC_UDLS` When defined, `` does not include `` with the user-defined string literals diff --git a/docs/mkdocs/docs/integration/cmake.md b/docs/mkdocs/docs/integration/cmake.md index 71512cbd5..4160584d5 100644 --- a/docs/mkdocs/docs/integration/cmake.md +++ b/docs/mkdocs/docs/integration/cmake.md @@ -169,6 +169,12 @@ Enable position diagnostics by defining macro [`JSON_DIAGNOSTIC_POSITIONS`](../a Disable default `enum` serialization by defining the macro [`JSON_DISABLE_ENUM_SERIALIZATION`](../api/macros/json_disable_enum_serialization.md). This option is `OFF` by default. +### `JSON_DisableTupleReferenceConversion` + +Disable the conversion from a one-element `std::tuple` holding a reference to a JSON value by defining the macro +[`JSON_DISABLE_TUPLE_REFERENCE_CONVERSION`](../api/macros/json_disable_tuple_reference_conversion.md). This option is +`OFF` by default. + ### `JSON_FastTests` Skip expensive/slow test suites. This option is `OFF` by default. Depends on `JSON_BuildTests`. diff --git a/docs/mkdocs/mkdocs.yml b/docs/mkdocs/mkdocs.yml index 21fad832a..b31fb1734 100644 --- a/docs/mkdocs/mkdocs.yml +++ b/docs/mkdocs/mkdocs.yml @@ -287,6 +287,7 @@ nav: - 'JSON_DIAGNOSTICS': api/macros/json_diagnostics.md - 'JSON_DIAGNOSTIC_POSITIONS': api/macros/json_diagnostic_positions.md - 'JSON_DISABLE_ENUM_SERIALIZATION': api/macros/json_disable_enum_serialization.md + - 'JSON_DISABLE_TUPLE_REFERENCE_CONVERSION': api/macros/json_disable_tuple_reference_conversion.md - 'JSON_HAS_CPP_11, JSON_HAS_CPP_14, JSON_HAS_CPP_17, JSON_HAS_CPP_20': api/macros/json_has_cpp_11.md - 'JSON_HAS_EXPERIMENTAL_FILESYSTEM, JSON_HAS_FILESYSTEM': api/macros/json_has_filesystem.md - 'JSON_HAS_RANGES': api/macros/json_has_ranges.md diff --git a/include/nlohmann/detail/conversions/to_json.hpp b/include/nlohmann/detail/conversions/to_json.hpp index fb48e4f82..92574c8ce 100644 --- a/include/nlohmann/detail/conversions/to_json.hpp +++ b/include/nlohmann/detail/conversions/to_json.hpp @@ -471,11 +471,13 @@ inline void to_json_tuple_impl(BasicJsonType& j, const Tuple& t, index_sequence< j = { std::get(t)... }; } -#if JSON_BRACE_INIT_COPY_SEMANTICS -// JSON_BRACE_INIT_COPY_SEMANTICS makes a one-element braced list copy its -// element instead of wrapping it, which would serialize std::tuple{5} as 5 -// rather than [5]. Build what the default deduction builds instead: an object -// if the element is a [string, value] pair, a one-element array otherwise. +// A one-element braced list does not reliably wrap its element: with +// JSON_BRACE_INIT_COPY_SEMANTICS it copies it, which would serialize +// std::tuple{5} as 5 rather than [5], and some compilers (e.g., Apple clang +// 15 and 16) copy an element that is itself a basic_json even without it, so +// std::tuple{true} became true rather than [true]. Build what the default +// deduction builds instead: an object if the element is a [string, value] pair, +// a one-element array otherwise. template inline void to_json_tuple_impl(BasicJsonType& j, const Tuple& t, index_sequence<0> /*unused*/) { @@ -493,7 +495,6 @@ inline void to_json_tuple_impl(BasicJsonType& j, const Tuple& t, index_sequence< j = BasicJsonType::array({std::move(element)}); } } -#endif template inline void to_json_tuple_impl(BasicJsonType& j, const Tuple& /*unused*/, index_sequence<> /*unused*/) diff --git a/include/nlohmann/detail/macro_scope.hpp b/include/nlohmann/detail/macro_scope.hpp index 0d524c710..2d0fccf18 100644 --- a/include/nlohmann/detail/macro_scope.hpp +++ b/include/nlohmann/detail/macro_scope.hpp @@ -915,3 +915,7 @@ void templated_json_throw(ExceptionType exception) #ifndef JSON_DISABLE_ENUM_SERIALIZATION #define JSON_DISABLE_ENUM_SERIALIZATION 0 #endif + +#ifndef JSON_DISABLE_TUPLE_REFERENCE_CONVERSION + #define JSON_DISABLE_TUPLE_REFERENCE_CONVERSION 0 +#endif diff --git a/include/nlohmann/detail/macro_unscope.hpp b/include/nlohmann/detail/macro_unscope.hpp index d2675d3f1..d6aa831d7 100644 --- a/include/nlohmann/detail/macro_unscope.hpp +++ b/include/nlohmann/detail/macro_unscope.hpp @@ -25,6 +25,7 @@ #undef JSON_INLINE_VARIABLE #undef JSON_NO_UNIQUE_ADDRESS #undef JSON_DISABLE_ENUM_SERIALIZATION +#undef JSON_DISABLE_TUPLE_REFERENCE_CONVERSION #ifndef JSON_TEST_KEEP_MACROS #undef JSON_CATCH diff --git a/include/nlohmann/detail/meta/type_traits.hpp b/include/nlohmann/detail/meta/type_traits.hpp index 36573bf6f..96b70a774 100644 --- a/include/nlohmann/detail/meta/type_traits.hpp +++ b/include/nlohmann/detail/meta/type_traits.hpp @@ -636,6 +636,18 @@ template struct is_compatible_type : is_compatible_type_impl {}; +// a one-element std::tuple holding a reference to BasicJsonType, as created by +// std::forward_as_tuple(j); see JSON_DISABLE_TUPLE_REFERENCE_CONVERSION +template +struct is_basic_json_reference_tuple : std::false_type {}; + +template +struct is_basic_json_reference_tuple> +{ + static constexpr bool value = + std::is_reference::value && std::is_same, BasicJsonType>::value; +}; + template struct is_compatible_binary_type { diff --git a/include/nlohmann/json.hpp b/include/nlohmann/json.hpp index 78c4e4f62..a82c1eb11 100644 --- a/include/nlohmann/json.hpp +++ b/include/nlohmann/json.hpp @@ -1596,7 +1596,12 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec template < typename CompatibleType, typename U = detail::uncvref_t, detail::enable_if_t < - !detail::is_basic_json::value && detail::is_compatible_type::value, int > = 0 > + !detail::is_basic_json::value && detail::is_compatible_type::value +#if JSON_DISABLE_TUPLE_REFERENCE_CONVERSION + // see https://github.com/nlohmann/json/issues/2226 + && !detail::is_basic_json_reference_tuple::value +#endif + , int > = 0 > basic_json(CompatibleType && val) noexcept(noexcept( // NOLINT(bugprone-forwarding-reference-overload,bugprone-exception-escape) JSONSerializer::to_json(std::declval(), std::forward(val)))) diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 5ef598150..d6d2ef347 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -3329,6 +3329,10 @@ void templated_json_throw(ExceptionType exception) #define JSON_DISABLE_ENUM_SERIALIZATION 0 #endif +#ifndef JSON_DISABLE_TUPLE_REFERENCE_CONVERSION + #define JSON_DISABLE_TUPLE_REFERENCE_CONVERSION 0 +#endif + #if JSON_HAS_THREE_WAY_COMPARISON #include // partial_ordering #endif @@ -4644,6 +4648,18 @@ template struct is_compatible_type : is_compatible_type_impl {}; +// a one-element std::tuple holding a reference to BasicJsonType, as created by +// std::forward_as_tuple(j); see JSON_DISABLE_TUPLE_REFERENCE_CONVERSION +template +struct is_basic_json_reference_tuple : std::false_type {}; + +template +struct is_basic_json_reference_tuple> +{ + static constexpr bool value = + std::is_reference::value && std::is_same, BasicJsonType>::value; +}; + template struct is_compatible_binary_type { @@ -7002,11 +7018,13 @@ inline void to_json_tuple_impl(BasicJsonType& j, const Tuple& t, index_sequence< j = { std::get(t)... }; } -#if JSON_BRACE_INIT_COPY_SEMANTICS -// JSON_BRACE_INIT_COPY_SEMANTICS makes a one-element braced list copy its -// element instead of wrapping it, which would serialize std::tuple{5} as 5 -// rather than [5]. Build what the default deduction builds instead: an object -// if the element is a [string, value] pair, a one-element array otherwise. +// A one-element braced list does not reliably wrap its element: with +// JSON_BRACE_INIT_COPY_SEMANTICS it copies it, which would serialize +// std::tuple{5} as 5 rather than [5], and some compilers (e.g., Apple clang +// 15 and 16) copy an element that is itself a basic_json even without it, so +// std::tuple{true} became true rather than [true]. Build what the default +// deduction builds instead: an object if the element is a [string, value] pair, +// a one-element array otherwise. template inline void to_json_tuple_impl(BasicJsonType& j, const Tuple& t, index_sequence<0> /*unused*/) { @@ -7024,7 +7042,6 @@ inline void to_json_tuple_impl(BasicJsonType& j, const Tuple& t, index_sequence< j = BasicJsonType::array({std::move(element)}); } } -#endif template inline void to_json_tuple_impl(BasicJsonType& j, const Tuple& /*unused*/, index_sequence<> /*unused*/) @@ -28523,7 +28540,12 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec template < typename CompatibleType, typename U = detail::uncvref_t, detail::enable_if_t < - !detail::is_basic_json::value && detail::is_compatible_type::value, int > = 0 > + !detail::is_basic_json::value && detail::is_compatible_type::value +#if JSON_DISABLE_TUPLE_REFERENCE_CONVERSION + // see https://github.com/nlohmann/json/issues/2226 + && !detail::is_basic_json_reference_tuple::value +#endif + , int > = 0 > basic_json(CompatibleType && val) noexcept(noexcept( // NOLINT(bugprone-forwarding-reference-overload,bugprone-exception-escape) JSONSerializer::to_json(std::declval(), std::forward(val)))) @@ -33572,6 +33594,7 @@ struct formatter // NOLINT(cert-dcl58-c #undef JSON_INLINE_VARIABLE #undef JSON_NO_UNIQUE_ADDRESS #undef JSON_DISABLE_ENUM_SERIALIZATION +#undef JSON_DISABLE_TUPLE_REFERENCE_CONVERSION #ifndef JSON_TEST_KEEP_MACROS #undef JSON_CATCH diff --git a/tests/src/unit-disable-tuple-reference-conversion.cpp b/tests/src/unit-disable-tuple-reference-conversion.cpp new file mode 100644 index 000000000..6d978620b --- /dev/null +++ b/tests/src/unit-disable-tuple-reference-conversion.cpp @@ -0,0 +1,93 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#include "doctest_compatibility.h" + +// This file tests the opt-in JSON_DISABLE_TUPLE_REFERENCE_CONVERSION, so it +// defines the macro itself rather than relying on a -D flag, and runs in every +// build. +#ifdef JSON_DISABLE_TUPLE_REFERENCE_CONVERSION + #undef JSON_DISABLE_TUPLE_REFERENCE_CONVERSION +#endif + +#define JSON_DISABLE_TUPLE_REFERENCE_CONVERSION 1 + +#include +using nlohmann::json; +using nlohmann::ordered_json; + +#include +#include +#include +#include + +// clang before 4 and GCC before 5 cannot create a std::tuple of basic_json +// references at all, with or without JSON_DISABLE_TUPLE_REFERENCE_CONVERSION: +// the tuple constructors make them instantiate basic_json's conversion operator +// for libstdc++'s internal tuple bases, which fails hard +#if (defined(__clang__) && __clang_major__ < 4) || (!defined(__clang__) && defined(__GNUC__) && __GNUC__ < 5) + #define SKIP_TESTS_FOR_JSON_REFERENCE_TUPLES +#endif + +TEST_CASE("JSON_DISABLE_TUPLE_REFERENCE_CONVERSION") +{ + SECTION("json is not constructible from a one-element tuple of a json reference") + { + CHECK_FALSE(std::is_constructible>::value); + CHECK_FALSE(std::is_constructible>::value); + CHECK_FALSE(std::is_constructible < json, std::tuple < json && >>::value); + CHECK_FALSE(std::is_constructible&>::value); + CHECK_FALSE(std::is_constructible>::value); + } + +#ifndef SKIP_TESTS_FOR_JSON_REFERENCE_TUPLES + SECTION("issue #2226 - tuple from tuple keeps the reference") + { + json j = true; + const std::tuple tup(std::forward_as_tuple(j)); + CHECK(&std::get<0>(tup) == &j); + } + + SECTION("tuple from tuple copies the element") + { + const json j = {{"key", "value"}}; + const std::tuple t1(std::forward_as_tuple(j)); + CHECK(std::get<0>(t1) == j); + + json j2 = "text"; + const std::tuple t2(std::forward_as_tuple(std::move(j2))); + CHECK(std::get<0>(t2) == "text"); + } +#endif + + SECTION("other tuple conversions are not affected") + { + const json j = true; + + // one-element tuple holding a json value + CHECK(json(std::make_tuple(j)) == json::array({true})); + + // tuples with more than one element, even when holding references + int i = 1; +#ifndef SKIP_TESTS_FOR_JSON_REFERENCE_TUPLES + CHECK(json(std::forward_as_tuple(i, j)) == json::array({1, true})); + CHECK(json(std::forward_as_tuple(j, j)) == json::array({true, true})); +#endif + + // one-element tuples holding references to other types + std::string s = "text"; + CHECK(json(std::forward_as_tuple(s)) == json::array({"text"})); + CHECK(json(std::forward_as_tuple(i)) == json::array({1})); + +#ifndef SKIP_TESTS_FOR_JSON_REFERENCE_TUPLES + // a reference to a different basic_json specialization + ordered_json oj = true; + CHECK(json(std::forward_as_tuple(oj)) == json::array({true})); +#endif + } +} diff --git a/tests/src/unit-regression2.cpp b/tests/src/unit-regression2.cpp index c466958bf..6446f5f8c 100644 --- a/tests/src/unit-regression2.cpp +++ b/tests/src/unit-regression2.cpp @@ -18,6 +18,19 @@ // for some reason including this after the json header leads to linker errors with VS 2017... #include +// skip tests if JSON_DISABLE_TUPLE_REFERENCE_CONVERSION=1 (#2226) +#if defined(JSON_DISABLE_TUPLE_REFERENCE_CONVERSION) && (JSON_DISABLE_TUPLE_REFERENCE_CONVERSION == 1) + #define SKIP_TESTS_FOR_TUPLE_REFERENCE_CONVERSION +#endif + +// clang before 4 and GCC before 5 cannot create a std::tuple of basic_json +// references at all, with or without JSON_DISABLE_TUPLE_REFERENCE_CONVERSION: +// the tuple constructors make them instantiate basic_json's conversion operator +// for libstdc++'s internal tuple bases, which fails hard +#if (defined(__clang__) && __clang_major__ < 4) || (!defined(__clang__) && defined(__GNUC__) && __GNUC__ < 5) + #define SKIP_TESTS_FOR_JSON_REFERENCE_TUPLES +#endif + #define JSON_TESTS_PRIVATE #include using json = nlohmann::json; @@ -28,6 +41,7 @@ using ordered_json = nlohmann::ordered_json; #include #include +#include #include #include @@ -542,6 +556,20 @@ TEST_CASE("regression tests 2") ))); } +#ifndef SKIP_TESTS_FOR_TUPLE_REFERENCE_CONVERSION + SECTION("issue #2226 - std::tuple dangling reference - implicit conversion") + { + // by default, a one-element tuple holding a json reference converts to + // a one-element array; JSON_DISABLE_TUPLE_REFERENCE_CONVERSION removes + // this conversion (see unit-disable-tuple-reference-conversion.cpp) + const json j = true; + CHECK(std::is_constructible>::value); +#ifndef SKIP_TESTS_FOR_JSON_REFERENCE_TUPLES + CHECK(json(std::forward_as_tuple(j)) == json::array({true})); +#endif + } +#endif + SECTION("PR #2181 - regression bug with lvalue") { // see https://github.com/nlohmann/json/pull/2181#issuecomment-653326060 From 7d7055ec500a519b0497f1c0aeefa977dcbfc0ab Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Wed, 30 Sep 2026 21:36:04 +0200 Subject: [PATCH 4/7] Fix stack overflow converting deep values between specializations (#5723) * Fix stack overflow converting deep values between specializations Constructing a basic_json from another specialization (json to ordered_json or back, also via get()) converted every container with its range constructor, which calls the converting constructor for each element. The call stack therefore grew with every nesting level, and a value nested some 30,000 levels deep overflowed it. The conversion now bounds its descent the way the copy constructor does since #5387: the first 128 levels are converted exactly as before, and below that convert_iteratively() finishes the value with an explicit stack. It builds each container bottom-up from its converted elements with the container's range constructor, so member order and keys that become equal are handled as before, and it gives a value its type only once its container exists, so an exception leaves nothing behind that cannot be destroyed. Parents (JSON_DIAGNOSTICS) and positions (JSON_DIAGNOSTIC_POSITIONS) are set for every value. Converting a null value no longer resets its positions: the constructor assigned null to a value that already was null, which swapped in the positions of the temporary. Fixes #5650. Signed-off-by: Niels Lohmann * Explain why converting null keeps positions and why next is a reference Review feedback on #5723 (gregmarr): clarify in comments that the converting constructor has already copied the positions of val, which the null case keeps like every other case, and that next must be a reference into pending so that ++next advances the stored iterator. Comments only; no code change. Signed-off-by: Niels Lohmann * Refer to recursion_depth_limit() in the convert_structured() docs The comment still named nesting_depth_limit, which #5637 removed on develop in favor of detail::recursion_depth_limit(). Signed-off-by: Niels Lohmann * Advance the pending iterator through pending.back() and shorten the null comment Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- include/nlohmann/json.hpp | 274 ++++++++++++++++++++---- single_include/nlohmann/json.hpp | 274 ++++++++++++++++++++---- tests/src/unit-allocator.cpp | 71 ++++++ tests/src/unit-diagnostic-positions.cpp | 52 +++++ tests/src/unit-diagnostics.cpp | 30 +++ tests/src/unit-large_json.cpp | 128 +++++++++++ 6 files changed, 735 insertions(+), 94 deletions(-) diff --git a/include/nlohmann/json.hpp b/include/nlohmann/json.hpp index a82c1eb11..344c9209b 100644 --- a/include/nlohmann/json.hpp +++ b/include/nlohmann/json.hpp @@ -906,11 +906,11 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /*! @brief how many levels the operation going on in this thread has descended into - Copying a value and comparing two values share this count. The library never - nests one inside the other - copying a value does not compare one, and - comparing two values does not copy them - and where user code nests them - anyway, sharing the count only ends a descent sooner than it had to, which - costs a little speed and is never wrong. + Copying a value, converting one from another specialization, and comparing + two values share this count. The library never nests one of them inside + another - none of them does either of the other two on the way - and where + user code nests them anyway, sharing the count only ends a descent sooner + than it had to, which costs a little speed and is never wrong. A byte is enough: the count never exceeds the limit by more than the single level that notices the limit has been reached. @@ -1267,6 +1267,222 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec copy_iteratively(src); } + /*! + @brief convert the value @a val of another specialization into this null + value; @a val must be neither an object nor an array + + Converting such a value never descends, so both ways of converting an + object or an array (@ref convert_structured) leave their elements of this + kind to the converting constructor, which leaves them to this. + */ + template + void convert_leaf(const BasicJsonType& val) + { + using other_boolean_t = typename BasicJsonType::boolean_t; + using other_number_float_t = typename BasicJsonType::number_float_t; + using other_number_integer_t = typename BasicJsonType::number_integer_t; + using other_number_unsigned_t = typename BasicJsonType::number_unsigned_t; + using other_string_t = typename BasicJsonType::string_t; + using other_binary_t = typename BasicJsonType::binary_t; + + switch (val.type()) + { + case value_t::boolean: + JSONSerializer::to_json(*this, val.template get()); + break; + case value_t::number_float: + JSONSerializer::to_json(*this, val.template get()); + break; + case value_t::number_integer: + JSONSerializer::to_json(*this, val.template get()); + break; + case value_t::number_unsigned: + JSONSerializer::to_json(*this, val.template get()); + break; + case value_t::string: + JSONSerializer::to_json(*this, val.template get_ref()); + break; + case value_t::binary: + JSONSerializer::to_json(*this, val.template get_ref()); + break; + case value_t::null: + // m_data.m_type is already value_t::null + break; + case value_t::discarded: + m_data.m_type = value_t::discarded; + break; + case value_t::object: // LCOV_EXCL_LINE + case value_t::array: // LCOV_EXCL_LINE + default: // LCOV_EXCL_LINE + JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE + } + } + + /// scratch space for the converted elements of the arrays that + /// @ref convert_iteratively has yet to create + using convert_scratch_t = std::vector>; + + /*! + @brief create the object or array @a val converted into this null value + + Its converted elements are the last `val.size()` entries of @a elements (an + array) or of @a members (an object); they are moved into the container in + one go and then removed. + */ + template + void convert_level(const BasicJsonType& val, convert_scratch_t& elements, copy_scratch_t& members) + { + if (val.is_object()) + { + const auto first = members.end() - static_cast(val.size()); + m_data.m_value.object = create(std::make_move_iterator(first), + std::make_move_iterator(members.end())); + // only now that the object exists may this stop being a null value + m_data.m_type = value_t::object; + members.erase(first, members.end()); + } + else + { + const auto first = elements.end() - static_cast(val.size()); + m_data.m_value.array = create(std::make_move_iterator(first), + std::make_move_iterator(elements.end())); + // only now that the array exists may this stop being a null value + m_data.m_type = value_t::array; + elements.erase(first, elements.end()); + } + + set_parents(); + } + + /*! + @brief convert the object or array @a val of another specialization into + this null value without recursing + + The containers whose conversion has begun are kept on an explicit stack + rather than on the call stack. Unlike @ref copy_iteratively, this builds + every container from the bottom up: all its elements are converted first, + and the container is then created from them in one go, the way the range + constructor that converts the levels above the bound does. The two object + types need not enumerate their members in the same order, so the members + could not be paired up by position anyway, and building from a range keeps + what the range constructor does with keys that become equal on conversion. + + Every value is complete before it is handed on, and a container gets its + type only once it exists, so whatever throws, every value left behind can + be destroyed. + */ + template + void convert_iteratively(const BasicJsonType& val) + { + using other_const_iterator = typename BasicJsonType::const_iterator; + + // the containers whose conversion has begun, innermost last, each with + // its element to convert next + std::vector> pending; + + // the converted elements of the pending arrays and the converted + // members of the pending objects, those of the innermost one last + convert_scratch_t elements; + copy_scratch_t members; + + pending.emplace_back(&val, val.cbegin()); + + for (;;) + { + const BasicJsonType& container = *pending.back().first; + // a copy, as descending below can reallocate pending; the + // iterator kept in pending is only advanced through pending.back() + const other_const_iterator next = pending.back().second; + + if (next != container.cend()) + { + if (next->is_structured()) + { + // convert its elements first; next stays where it is until + // the converted container is handed back to this one + pending.emplace_back(&*next, next->cbegin()); + continue; + } + + // the converting constructor does not descend into this value + if (container.is_object()) + { + members.emplace_back(next.key(), *next); + } + else + { + elements.emplace_back(*next); + } + ++pending.back().second; + continue; + } + + // all elements of the container are converted: create it + pending.pop_back(); + + if (pending.empty()) + { + convert_level(container, elements, members); + return; + } + + basic_json converted; + converted.convert_level(container, elements, members); +#if JSON_DIAGNOSTIC_POSITIONS + converted.start_position = container.start_pos(); + converted.end_position = container.end_pos(); +#endif + + // hand it to the container it is an element of + if (pending.back().first->is_object()) + { + members.emplace_back(pending.back().second.key(), std::move(converted)); + } + else + { + elements.push_back(std::move(converted)); + } + ++pending.back().second; + } + } + + /*! + @brief convert the object or array @a val of another specialization into + this null value + + Converting a container converts its elements, so a value nested deeply + enough used to exhaust the call stack. The descent is bounded here as in + @ref copy_structured: the first `detail::recursion_depth_limit()` levels + are converted by the containers' range constructors, just as they always were, + and anything below that is converted without the call stack by + @ref convert_iteratively. + + @sa https://github.com/nlohmann/json/issues/5650 + */ + template + void convert_structured(const BasicJsonType& val) + { + const nesting_depth_guard guard; + + if (JSON_HEDLEY_LIKELY(guard.okay())) + { + // every element comes back to the converting constructor + if (val.is_object()) + { + using other_object_t = typename BasicJsonType::object_t; + JSONSerializer::to_json(*this, val.template get_ref()); + } + else + { + using other_array_t = typename BasicJsonType::array_t; + JSONSerializer::to_json(*this, val.template get_ref()); + } + return; + } + + convert_iteratively(val); + } + /// the result of comparing two values, including values that cannot be /// ordered at all, such as a discarded value or a NaN @@ -1622,49 +1838,13 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec end_position(val.end_pos()) #endif { - using other_boolean_t = typename BasicJsonType::boolean_t; - using other_number_float_t = typename BasicJsonType::number_float_t; - using other_number_integer_t = typename BasicJsonType::number_integer_t; - using other_number_unsigned_t = typename BasicJsonType::number_unsigned_t; - using other_string_t = typename BasicJsonType::string_t; - using other_object_t = typename BasicJsonType::object_t; - using other_array_t = typename BasicJsonType::array_t; - using other_binary_t = typename BasicJsonType::binary_t; - - switch (val.type()) + if (val.is_structured()) { - case value_t::boolean: - JSONSerializer::to_json(*this, val.template get()); - break; - case value_t::number_float: - JSONSerializer::to_json(*this, val.template get()); - break; - case value_t::number_integer: - JSONSerializer::to_json(*this, val.template get()); - break; - case value_t::number_unsigned: - JSONSerializer::to_json(*this, val.template get()); - break; - case value_t::string: - JSONSerializer::to_json(*this, val.template get_ref()); - break; - case value_t::object: - JSONSerializer::to_json(*this, val.template get_ref()); - break; - case value_t::array: - JSONSerializer::to_json(*this, val.template get_ref()); - break; - case value_t::binary: - JSONSerializer::to_json(*this, val.template get_ref()); - break; - case value_t::null: - *this = nullptr; - break; - case value_t::discarded: - m_data.m_type = value_t::discarded; - break; - default: // LCOV_EXCL_LINE - JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE + convert_structured(val); + } + else + { + convert_leaf(val); } JSON_ASSERT(m_data.m_type == val.type()); diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index d6d2ef347..1fa63f344 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -27850,11 +27850,11 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /*! @brief how many levels the operation going on in this thread has descended into - Copying a value and comparing two values share this count. The library never - nests one inside the other - copying a value does not compare one, and - comparing two values does not copy them - and where user code nests them - anyway, sharing the count only ends a descent sooner than it had to, which - costs a little speed and is never wrong. + Copying a value, converting one from another specialization, and comparing + two values share this count. The library never nests one of them inside + another - none of them does either of the other two on the way - and where + user code nests them anyway, sharing the count only ends a descent sooner + than it had to, which costs a little speed and is never wrong. A byte is enough: the count never exceeds the limit by more than the single level that notices the limit has been reached. @@ -28211,6 +28211,222 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec copy_iteratively(src); } + /*! + @brief convert the value @a val of another specialization into this null + value; @a val must be neither an object nor an array + + Converting such a value never descends, so both ways of converting an + object or an array (@ref convert_structured) leave their elements of this + kind to the converting constructor, which leaves them to this. + */ + template + void convert_leaf(const BasicJsonType& val) + { + using other_boolean_t = typename BasicJsonType::boolean_t; + using other_number_float_t = typename BasicJsonType::number_float_t; + using other_number_integer_t = typename BasicJsonType::number_integer_t; + using other_number_unsigned_t = typename BasicJsonType::number_unsigned_t; + using other_string_t = typename BasicJsonType::string_t; + using other_binary_t = typename BasicJsonType::binary_t; + + switch (val.type()) + { + case value_t::boolean: + JSONSerializer::to_json(*this, val.template get()); + break; + case value_t::number_float: + JSONSerializer::to_json(*this, val.template get()); + break; + case value_t::number_integer: + JSONSerializer::to_json(*this, val.template get()); + break; + case value_t::number_unsigned: + JSONSerializer::to_json(*this, val.template get()); + break; + case value_t::string: + JSONSerializer::to_json(*this, val.template get_ref()); + break; + case value_t::binary: + JSONSerializer::to_json(*this, val.template get_ref()); + break; + case value_t::null: + // m_data.m_type is already value_t::null + break; + case value_t::discarded: + m_data.m_type = value_t::discarded; + break; + case value_t::object: // LCOV_EXCL_LINE + case value_t::array: // LCOV_EXCL_LINE + default: // LCOV_EXCL_LINE + JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE + } + } + + /// scratch space for the converted elements of the arrays that + /// @ref convert_iteratively has yet to create + using convert_scratch_t = std::vector>; + + /*! + @brief create the object or array @a val converted into this null value + + Its converted elements are the last `val.size()` entries of @a elements (an + array) or of @a members (an object); they are moved into the container in + one go and then removed. + */ + template + void convert_level(const BasicJsonType& val, convert_scratch_t& elements, copy_scratch_t& members) + { + if (val.is_object()) + { + const auto first = members.end() - static_cast(val.size()); + m_data.m_value.object = create(std::make_move_iterator(first), + std::make_move_iterator(members.end())); + // only now that the object exists may this stop being a null value + m_data.m_type = value_t::object; + members.erase(first, members.end()); + } + else + { + const auto first = elements.end() - static_cast(val.size()); + m_data.m_value.array = create(std::make_move_iterator(first), + std::make_move_iterator(elements.end())); + // only now that the array exists may this stop being a null value + m_data.m_type = value_t::array; + elements.erase(first, elements.end()); + } + + set_parents(); + } + + /*! + @brief convert the object or array @a val of another specialization into + this null value without recursing + + The containers whose conversion has begun are kept on an explicit stack + rather than on the call stack. Unlike @ref copy_iteratively, this builds + every container from the bottom up: all its elements are converted first, + and the container is then created from them in one go, the way the range + constructor that converts the levels above the bound does. The two object + types need not enumerate their members in the same order, so the members + could not be paired up by position anyway, and building from a range keeps + what the range constructor does with keys that become equal on conversion. + + Every value is complete before it is handed on, and a container gets its + type only once it exists, so whatever throws, every value left behind can + be destroyed. + */ + template + void convert_iteratively(const BasicJsonType& val) + { + using other_const_iterator = typename BasicJsonType::const_iterator; + + // the containers whose conversion has begun, innermost last, each with + // its element to convert next + std::vector> pending; + + // the converted elements of the pending arrays and the converted + // members of the pending objects, those of the innermost one last + convert_scratch_t elements; + copy_scratch_t members; + + pending.emplace_back(&val, val.cbegin()); + + for (;;) + { + const BasicJsonType& container = *pending.back().first; + // a copy, as descending below can reallocate pending; the + // iterator kept in pending is only advanced through pending.back() + const other_const_iterator next = pending.back().second; + + if (next != container.cend()) + { + if (next->is_structured()) + { + // convert its elements first; next stays where it is until + // the converted container is handed back to this one + pending.emplace_back(&*next, next->cbegin()); + continue; + } + + // the converting constructor does not descend into this value + if (container.is_object()) + { + members.emplace_back(next.key(), *next); + } + else + { + elements.emplace_back(*next); + } + ++pending.back().second; + continue; + } + + // all elements of the container are converted: create it + pending.pop_back(); + + if (pending.empty()) + { + convert_level(container, elements, members); + return; + } + + basic_json converted; + converted.convert_level(container, elements, members); +#if JSON_DIAGNOSTIC_POSITIONS + converted.start_position = container.start_pos(); + converted.end_position = container.end_pos(); +#endif + + // hand it to the container it is an element of + if (pending.back().first->is_object()) + { + members.emplace_back(pending.back().second.key(), std::move(converted)); + } + else + { + elements.push_back(std::move(converted)); + } + ++pending.back().second; + } + } + + /*! + @brief convert the object or array @a val of another specialization into + this null value + + Converting a container converts its elements, so a value nested deeply + enough used to exhaust the call stack. The descent is bounded here as in + @ref copy_structured: the first `detail::recursion_depth_limit()` levels + are converted by the containers' range constructors, just as they always were, + and anything below that is converted without the call stack by + @ref convert_iteratively. + + @sa https://github.com/nlohmann/json/issues/5650 + */ + template + void convert_structured(const BasicJsonType& val) + { + const nesting_depth_guard guard; + + if (JSON_HEDLEY_LIKELY(guard.okay())) + { + // every element comes back to the converting constructor + if (val.is_object()) + { + using other_object_t = typename BasicJsonType::object_t; + JSONSerializer::to_json(*this, val.template get_ref()); + } + else + { + using other_array_t = typename BasicJsonType::array_t; + JSONSerializer::to_json(*this, val.template get_ref()); + } + return; + } + + convert_iteratively(val); + } + /// the result of comparing two values, including values that cannot be /// ordered at all, such as a discarded value or a NaN @@ -28566,49 +28782,13 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec end_position(val.end_pos()) #endif { - using other_boolean_t = typename BasicJsonType::boolean_t; - using other_number_float_t = typename BasicJsonType::number_float_t; - using other_number_integer_t = typename BasicJsonType::number_integer_t; - using other_number_unsigned_t = typename BasicJsonType::number_unsigned_t; - using other_string_t = typename BasicJsonType::string_t; - using other_object_t = typename BasicJsonType::object_t; - using other_array_t = typename BasicJsonType::array_t; - using other_binary_t = typename BasicJsonType::binary_t; - - switch (val.type()) + if (val.is_structured()) { - case value_t::boolean: - JSONSerializer::to_json(*this, val.template get()); - break; - case value_t::number_float: - JSONSerializer::to_json(*this, val.template get()); - break; - case value_t::number_integer: - JSONSerializer::to_json(*this, val.template get()); - break; - case value_t::number_unsigned: - JSONSerializer::to_json(*this, val.template get()); - break; - case value_t::string: - JSONSerializer::to_json(*this, val.template get_ref()); - break; - case value_t::object: - JSONSerializer::to_json(*this, val.template get_ref()); - break; - case value_t::array: - JSONSerializer::to_json(*this, val.template get_ref()); - break; - case value_t::binary: - JSONSerializer::to_json(*this, val.template get_ref()); - break; - case value_t::null: - *this = nullptr; - break; - case value_t::discarded: - m_data.m_type = value_t::discarded; - break; - default: // LCOV_EXCL_LINE - JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE + convert_structured(val); + } + else + { + convert_leaf(val); } JSON_ASSERT(m_data.m_type == val.type()); diff --git a/tests/src/unit-allocator.cpp b/tests/src/unit-allocator.cpp index ce498558d..151c268d7 100644 --- a/tests/src/unit-allocator.cpp +++ b/tests/src/unit-allocator.cpp @@ -479,6 +479,77 @@ TEST_CASE("deep copy uses the provided allocator") CHECK(copy == j); } +namespace +{ +// the number of constructions countdown_allocator lets happen, including the +// one that fails; 0 means none ever fails +std::size_t constructions_until_failure = 0; + +template +struct countdown_allocator : std::allocator +{ + using std::allocator::allocator; + + template + void construct(U* p, Args&& ... args) + { + if (constructions_until_failure != 0 && --constructions_until_failure == 0) + { + throw std::bad_alloc(); + } + + ::new (static_cast(p)) U(std::forward(args)...); + } + + template + struct rebind + { + using other = countdown_allocator; + }; +}; +} // namespace + +TEST_CASE("converting a deeply nested value from another specialization fails cleanly (#5650)") +{ + using countdown_json = nlohmann::basic_json; + + // deeper than the 128 levels the converting constructor descends into, so + // that failures land on both sides of the bound - or, built with + // JSON_NO_THREAD_LOCAL, all in the iterative conversion + json j = {1, "two", {{"three", 3}}}; + for (std::size_t i = 0; i < 150; ++i) + { + j = json{{"a", json::array({j, "sibling"})}}; + } + + // Fail every construction in turn. Each failure has to reach the caller, + // and everything built until then has to be destroyed cleanly. + std::size_t failures = 0; + for (std::size_t n = 1;; ++n) + { + constructions_until_failure = n; + try + { + const countdown_json converted = j; + constructions_until_failure = 0; + CHECK(converted.dump() == j.dump()); + break; + } + catch (const std::bad_alloc&) + { + ++failures; + } + } + CHECK(failures > 0); +} + namespace { template diff --git a/tests/src/unit-diagnostic-positions.cpp b/tests/src/unit-diagnostic-positions.cpp index 5326094c7..f5c6bc648 100644 --- a/tests/src/unit-diagnostic-positions.cpp +++ b/tests/src/unit-diagnostic-positions.cpp @@ -141,6 +141,58 @@ TEST_CASE("Better diagnostics with positions") check_objects(300); } + SECTION("converting keeps the positions of nested values (#5650)") + { + // Values nested deeper than the converting constructor's descent bound + // are converted without the call stack, on a path that has to carry the + // positions of every value over itself. Objects and arrays take turns, + // and the innermost value is null, which used to lose its positions. + const auto check_conversion = [](std::size_t depth) + { + CAPTURE(depth) + + std::string text; + std::string closing; + for (std::size_t i = 0; i < depth; ++i) + { + text += (i % 2 == 0) ? "[12, " : R"({"b":1, "a":)"; + closing += (i % 2 == 0) ? ']' : '}'; + } + text += "null"; + text.append(closing.rbegin(), closing.rend()); + + const json original = json::parse(text); + const nlohmann::ordered_json converted = original; + + const json* o = &original; + const nlohmann::ordered_json* c = &converted; + for (std::size_t level = 0; level <= depth; ++level) + { + CAPTURE(level) + REQUIRE(c->start_pos() == o->start_pos()); + REQUIRE(c->end_pos() == o->end_pos()); + + if (level < depth) + { + // the number beside the value nested next + const json& o_number = o->is_object() ? o->at("b") : o->at(0); + const nlohmann::ordered_json& c_number = c->is_object() ? c->at("b") : c->at(0); + REQUIRE(c_number.start_pos() == o_number.start_pos()); + REQUIRE(c_number.end_pos() == o_number.end_pos()); + + o = o->is_object() ? &o->at("a") : &o->at(1); + c = c->is_object() ? &c->at("a") : &c->at(1); + } + } + }; + + check_conversion(1); + check_conversion(127); + check_conversion(128); + check_conversion(129); + check_conversion(300); + } + SECTION("JSON patch add to primitive parent (#4292)") { // the JSON Patch "add" target /foo/bar/baz has a string parent diff --git a/tests/src/unit-diagnostics.cpp b/tests/src/unit-diagnostics.cpp index b3778e802..ee7360eb1 100644 --- a/tests/src/unit-diagnostics.cpp +++ b/tests/src/unit-diagnostics.cpp @@ -341,6 +341,36 @@ TEST_CASE("Regression tests for extended diagnostics") } } + SECTION("Regression test for issue #5650 - converting keeps the parents of nested values") + { + // A value nested deeper than the converting constructor's descent bound + // is converted without the call stack. Every container that path creates + // has to have the parents of its children set, or the JSON Pointer in the + // diagnostic is cut short. Objects and arrays take turns. + const std::size_t pairs = 150; + + json j = "not a number"; + std::string pointer; + for (std::size_t i = 0; i < pairs; ++i) + { + j = json{{"a", json::array({j})}}; + pointer += "/a/0"; + } + + const nlohmann::ordered_json converted = j; + + const nlohmann::ordered_json* inner = &converted; + for (std::size_t i = 0; i < pairs; ++i) + { + inner = &inner->at("a").at(0); + } + + std::string const expected = "[json.exception.type_error.302] (" + pointer + ") type must be number, but is string"; + int i = 0; + CHECK_THROWS_WITH_AS(i = inner->get(), expected.c_str(), nlohmann::ordered_json::type_error); + CHECK(i == 0); + } + SECTION("Regression test for issue #5668 - wrong path for std::map/unordered_map with non-string keys") { // a map with non-string keys is read from an array of [key, value] arrays; diff --git a/tests/src/unit-large_json.cpp b/tests/src/unit-large_json.cpp index 8204ed9b6..97f665848 100644 --- a/tests/src/unit-large_json.cpp +++ b/tests/src/unit-large_json.cpp @@ -13,6 +13,7 @@ using nlohmann::json; #include #include +#include TEST_CASE("tests on very large JSONs") { @@ -53,6 +54,24 @@ const json* innermost_value(const json& j, std::size_t& depth) return current; } +// The text of a value nested depth levels deep around the number 0. Level i is +// an array if pattern[i % pattern.size()] is '[', and otherwise an object with +// the single member "a", which every object type enumerates in the same order. +std::string nested_text(std::size_t depth, const std::string& pattern) +{ + std::string text; + std::string closing; + for (std::size_t i = 0; i < depth; ++i) + { + const bool array = pattern[i % pattern.size()] == '['; + text += array ? "[" : "{\"a\":"; + closing += array ? ']' : '}'; + } + text += '0'; + text.append(closing.rbegin(), closing.rend()); + return text; +} + } // namespace TEST_CASE("tests on deeply nested JSONs") @@ -224,5 +243,114 @@ TEST_CASE("tests on deeply nested JSONs") CHECK(*innermost_value(j, unused) == 0); } } + + SECTION("issue #5650 - stack overflow converting between specializations") + { + const std::vector patterns = {"[", "{", "[{"}; + + SECTION("json to ordered_json") + { + for (const auto& pattern : patterns) + { + CAPTURE(pattern); + const std::string text = nested_text(depth, pattern); + const json j = json::parse(text); + + const nlohmann::ordered_json converted = j; + CHECK(converted.dump() == text); + } + } + + SECTION("ordered_json to json") + { + for (const auto& pattern : patterns) + { + CAPTURE(pattern); + const std::string text = nested_text(depth, pattern); + const nlohmann::ordered_json o = nlohmann::ordered_json::parse(text); + + const json converted = o; + CHECK(converted.dump() == text); + } + } + + SECTION("get()") + { + for (const auto& pattern : patterns) + { + CAPTURE(pattern); + const std::string text = nested_text(depth, pattern); + const json j = json::parse(text); + + CHECK(j.get().dump() == text); + } + } + + SECTION("depths around the bound of the recursive descent") + { + for (std::size_t d = 1; d <= 300; ++d) + { + CAPTURE(d); + for (const auto& pattern : patterns) + { + CAPTURE(pattern); + const std::string text = nested_text(d, pattern); + const json j = json::parse(text); + + const nlohmann::ordered_json converted = j; + CHECK(converted.dump() == text); + const json back = converted; + CHECK(back.dump() == text); + } + } + } + + SECTION("values below the bound are converted as values above it") + { + // Bury a value below the bound, where it is converted without the + // call stack, and compare it with the same value converted on its + // own by the containers' range constructors. Its objects have + // members that the two object types enumerate in different orders. + const auto bury = [](nlohmann::ordered_json value) + { + for (std::size_t i = 0; i < 200; ++i) + { + value = nlohmann::ordered_json::array({std::move(value)}); + } + return value; + }; + const auto dig = [](const json & value) + { + const json* current = &value; + for (std::size_t i = 0; i < 200; ++i) + { + current = ¤t->at(0); + } + return current; + }; + + nlohmann::ordered_json value = nlohmann::ordered_json::object(); + value["z"] = {1, -2, 3U, 4.5, true, nullptr, "six", nlohmann::ordered_json::binary({7, 8}, 9), + nlohmann::ordered_json::binary({10}), nlohmann::ordered_json::array(), nlohmann::ordered_json::object() + }; + value["y"] = {{"x", {{"w", 1}, {"v", 2}}}, {"u", {3, {{"t", 4}, {"s", 5}}}}}; + value["r"] = nlohmann::ordered_json::array({nlohmann::ordered_json(nlohmann::ordered_json::value_t::discarded)}); + + const json converted_above = value; + const json buried = bury(value); + const json& converted_below = *dig(buried); + + CHECK(converted_below.dump() == converted_above.dump()); + CHECK(converted_below.at("z").at(7).get_binary().subtype() == 9); + CHECK_FALSE(converted_below.at("z").at(8).get_binary().has_subtype()); + CHECK(converted_below.at("r").at(0).is_discarded()); + + // a discarded value is never equal to anything, so compare the rest + value.erase("r"); + const json without_discarded_above = value; + const json without_discarded_buried = bury(value); + CHECK(*dig(without_discarded_buried) == without_discarded_above); + } + } } From 791cd88dfcf1f26a3e506559b065b9dd54e4181b Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Wed, 30 Sep 2026 22:15:20 +0200 Subject: [PATCH 5/7] Fix raw-TeX formulas and other documentation infrastructure debt (#5736) * Fix math formulas rendering as raw TeX in the published docs The privacy plugin self-hosts MathJax 2.7.0 but drops its ?config=TeX-MML-AM_CHTML query string, so the rehosted script loads no input jax and the 10 formulas across 6 pages render as raw TeX to readers. Remove pymdownx.arithmatex and the MathJax extra_javascript entry, and rewrite the formulas in plain HTML (, ) instead. This also drops a nine-year-old third-party script from every page. Part of #5718 Signed-off-by: Niels Lohmann * Fix publish_documentation triggers and persist unneeded git credentials publish_documentation.yml only triggered on docs/mkdocs/** pushes, but the site also embeds .github/CODE_OF_CONDUCT.md, CONTRIBUTING.md, SECURITY.md, cmake/{clang,gcc}_flags.cmake, .clang-tidy, tools/astyle/.astylerc and tests/fmt_formatter/project/main.cpp via pymdownx.snippets, so changes to those files never republished the site. Extend the path filter to cover them, and switch runs-on from the long-pinned ubuntu-22.04 to ubuntu-latest to match ci_test_documentation. Also add persist-credentials: false to the checkouts in ci_icpx and ci_nvhpc (ubuntu.yml) and msvc-vs2026/msvc-arm64 (windows.yml), none of which pushes with git, matching every other checkout in these workflows. Part of #5718 Signed-off-by: Niels Lohmann * Fix example build: broken debug echo, deprecations hidden for everything docs/Makefile's debug echo used a space instead of a comma in $(call cxx_standard ...), so it always printed an empty standard. Every example was also compiled with -Wno-deprecated-declarations, which would silently hide an accidental deprecated-API call in any of them. Factor the duplicated compile flags into EXAMPLE_CPPFLAGS/ EXAMPLE_WARNFLAGS, build with -Werror=deprecated-declarations by default, and only allow the three examples that intentionally document deprecated API (the DEPRECATED_EXAMPLES list) to suppress it. Part of #5718 Signed-off-by: Niels Lohmann * Fix check_structure.py NOLINT parsing and an off-by-one report line `line.strip("")` strips any of the characters in those sets from both ends, not a literal prefix; it only happened to work for "Examples". A NOLINT'd section name starting with N, O, L, I or T (e.g. "Notes", "Template parameters", "Iterator invalidation", "Literals") was silently mangled, so the suppression did not apply and the checker could report a spurious missing/misordered section. Parse the comment with a regex instead. The same fragile strip() pattern was used for heading text; replace it with a plain prefix slice. Also fix the admonition_title report, which used the 0-based line index while every other report in the file uses lineno+1. Part of #5718 Signed-off-by: Niels Lohmann * Fix docset README's non-existent make target and stale fallback URL The README told readers to run `make nlohmann_json.docset`, but the Makefile's targets are `all`, `JSON_for_Modern_C++.docset` and `install_docset_zeal`; the documented command has failed with "No rule to make target" since #2967 (2021). Point the README at the real target and folder name. Info.plist's DashDocSetFallbackURL also still pointed at the old nlohmann.github.io/json/ URL instead of the canonical https://json.nlohmann.me/ from mkdocs.yml's site_url. Leave list_missing_pages/list_removed_paths alone: they may become redundant once #5638's check_docset() lands, which is a follow-up. Part of #5718 Signed-off-by: Niels Lohmann * Remove two leftover Doxygen-era .link files from the examples directory parse__iterator_pair.link and parse__pointers.link each held only a Wandbox "online" permalink from the old Doxygen docs. #3071 deleted every other .link file in 2021; these two came in through a parallel PR (#3100) and were never referenced by any page, script or config. Part of #5718 Signed-off-by: Niels Lohmann * Fix MacPorts CMake example to include its own snippet files The MacPorts "Example: CMake" block included integration/homebrew/example.cpp and integration/homebrew/CMakeLists.txt instead of the MacPorts files right next to it, a copy-paste slip from the Homebrew section. Nothing referenced integration/macports/CMakeLists.txt as a result. The page rendered correctly only because the homebrew, macports and vcpkg/CMakeLists.txt snippets are byte-identical, so a future edit to the MacPorts files would not have shown up on the page. Part of #5718 item 6 Signed-off-by: Niels Lohmann * Do not persist git credentials in publish_documentation's checkout The checkout step in publish_documentation.yml left the default persist-credentials: true, so GITHUB_TOKEN stayed writable in .git/config for the rest of the job (zizmor's artipacked finding). The Deploy documentation step authenticates through its own github_token input to peaceiris/actions-gh-pages and does not push with the checked-out credentials, so persist-credentials: false is safe here, matching every other checkout in the workflow set. Overlaps #5638, which edits this same checkout step (adds fetch-depth: 0); expect a rebase conflict there. Part of #5718 item 2 Signed-off-by: Niels Lohmann * Replace list_missing_pages/list_removed_paths with a comm(1)-based diff The docset Makefile's list_missing_pages ran one sqlite3 query per mkdocs page, and list_removed_paths nested a loop over all mkdocs pages inside a loop over all docset index paths (O(n*m) shell iteration). Issue #5718 item 5 suggested removing or reducing these targets once #5638's check_docset() lands, but that PR is still open and covers only API pages and macros, not the full page set these targets check. Replace the loops with two sorted path lists (DOCSET_PAGE_PATHS from mkdocs' markdown sources, DOCSET_INDEX_PATHS from the built docset index) compared with a single comm(1) call each, verified to produce output identical to the old loops against the current docSet.dsidx. The sed expression used '#' as its delimiter, which GNU Make reads as a comment character even inside a variable assignment, truncating the line and orphaning the closing paren of $(shell ...) ("unterminated call to function 'shell': missing ')'"). Use '@' as the delimiter instead. Part of #5718 item 5 Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- .github/workflows/publish_documentation.yml | 14 +++++++- .github/workflows/ubuntu.yml | 4 +++ .github/workflows/windows.yml | 4 +++ docs/Makefile | 21 +++++++++--- docs/docset/Info.plist | 2 +- docs/docset/Makefile | 34 +++++++------------ docs/docset/README.md | 5 +-- .../docs/api/basic_json/number_integer_t.md | 2 +- .../docs/api/basic_json/number_unsigned_t.md | 2 +- .../mkdocs/docs/api/basic_json/parse_error.md | 2 +- .../docs/examples/parse__iterator_pair.link | 1 - .../mkdocs/docs/examples/parse__pointers.link | 1 - .../docs/features/binary_formats/bjdata.md | 2 +- docs/mkdocs/docs/features/types/index.md | 2 +- .../docs/features/types/number_handling.md | 8 ++--- .../docs/integration/package_managers.md | 4 +-- docs/mkdocs/mkdocs.yml | 4 --- docs/mkdocs/scripts/check_structure.py | 10 +++--- 18 files changed, 69 insertions(+), 53 deletions(-) delete mode 100644 docs/mkdocs/docs/examples/parse__iterator_pair.link delete mode 100644 docs/mkdocs/docs/examples/parse__pointers.link diff --git a/.github/workflows/publish_documentation.yml b/.github/workflows/publish_documentation.yml index 189d95419..f65d63a92 100644 --- a/.github/workflows/publish_documentation.yml +++ b/.github/workflows/publish_documentation.yml @@ -7,6 +7,16 @@ on: - develop paths: - docs/mkdocs/** + # the site also embeds these files via pymdownx.snippets + # (mkdocs.yml sets restrict_base_path: false for this) + - .clang-tidy + - .github/CODE_OF_CONDUCT.md + - .github/CONTRIBUTING.md + - .github/SECURITY.md + - cmake/clang_flags.cmake + - cmake/gcc_flags.cmake + - tests/fmt_formatter/project/main.cpp + - tools/astyle/.astylerc workflow_dispatch: # we don't want to have concurrent jobs, and we don't want to cancel running jobs to avoid broken publications @@ -23,7 +33,7 @@ jobs: contents: write if: github.repository == 'nlohmann/json' - runs-on: ubuntu-22.04 + runs-on: ubuntu-latest steps: - name: Harden Runner uses: step-security/harden-runner@e14015d583714f6e62063499dc959a02595150a1 # v2.21.1 @@ -31,6 +41,8 @@ jobs: egress-policy: audit - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false - name: Install virtual environment run: make install_venv -C docs/mkdocs diff --git a/.github/workflows/ubuntu.yml b/.github/workflows/ubuntu.yml index d8f15e3dd..fb3cd4148 100644 --- a/.github/workflows/ubuntu.yml +++ b/.github/workflows/ubuntu.yml @@ -346,6 +346,8 @@ jobs: container: intel/oneapi-hpckit:latest steps: - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false - name: Get latest CMake and ninja uses: lukka/get-cmake@fffaaafeea488556c2c12dad60690008bc1caacb # v4.4.2 - name: Run CMake @@ -358,6 +360,8 @@ jobs: container: nvcr.io/nvidia/nvhpc:25.5-devel-cuda12.9-ubuntu22.04 steps: - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false - name: Get latest CMake and ninja uses: lukka/get-cmake@fffaaafeea488556c2c12dad60690008bc1caacb # v4.4.2 - name: Run CMake diff --git a/.github/workflows/windows.yml b/.github/workflows/windows.yml index ea742773f..65169eac4 100644 --- a/.github/workflows/windows.yml +++ b/.github/workflows/windows.yml @@ -87,6 +87,8 @@ jobs: steps: - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false - name: Get latest CMake and ninja uses: lukka/get-cmake@fffaaafeea488556c2c12dad60690008bc1caacb # v4.4.2 - name: Set extra CXX_FLAGS for latest std_version @@ -123,6 +125,8 @@ jobs: steps: - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false - name: Run CMake (Release) run: cmake -S . -B build -G "Visual Studio 18 2026" -A ARM64 -DJSON_BuildTests=On -DCMAKE_CXX_FLAGS="/W4 /WX" if: matrix.build_type == 'Release' diff --git a/docs/Makefile b/docs/Makefile index 0412fb90a..a61c30182 100644 --- a/docs/Makefile +++ b/docs/Makefile @@ -11,20 +11,31 @@ EXAMPLES = $(wildcard mkdocs/docs/examples/*.cpp) cxx_standard = $(lastword c++11 $(filter c++%, $(subst ., ,$1))) +# common compile flags for the stand-alone example files +EXAMPLE_CPPFLAGS = -I $(SRCDIR) -DJSON_USE_GLOBAL_UDLS=0 +EXAMPLE_WARNFLAGS = -Werror=deprecated-declarations + +# examples that document deprecated API and are allowed to use it +DEPRECATED_EXAMPLES = $(addprefix mkdocs/docs/examples/, \ + json_pointer__operator__equal_stringtype \ + json_pointer__operator__notequal_stringtype \ + json_pointer__operator_string_t) +$(DEPRECATED_EXAMPLES:=.output) $(DEPRECATED_EXAMPLES:=.test): EXAMPLE_WARNFLAGS = -Wno-deprecated-declarations + # create output from a stand-alone example file %.output: %.cpp - @echo "standard $(call cxx_standard $(<:.cpp=))" + @echo "standard $(call cxx_standard,$(<:.cpp=))" $(MAKE) $(<:.cpp=) \ - CPPFLAGS="-I $(SRCDIR) -DJSON_USE_GLOBAL_UDLS=0" \ - CXXFLAGS="-std=$(call cxx_standard,$(<:.cpp=)) -Wno-deprecated-declarations" + CPPFLAGS="$(EXAMPLE_CPPFLAGS)" \ + CXXFLAGS="-std=$(call cxx_standard,$(<:.cpp=)) $(EXAMPLE_WARNFLAGS)" ./$(<:.cpp=) > $@ rm $(<:.cpp=) # compare created output with current output of the example files %.test: %.cpp $(MAKE) $(<:.cpp=) \ - CPPFLAGS="-I $(SRCDIR) -DJSON_USE_GLOBAL_UDLS=0" \ - CXXFLAGS="-std=$(call cxx_standard,$(<:.cpp=)) -Wno-deprecated-declarations" + CPPFLAGS="$(EXAMPLE_CPPFLAGS)" \ + CXXFLAGS="-std=$(call cxx_standard,$(<:.cpp=)) $(EXAMPLE_WARNFLAGS)" ./$(<:.cpp=) > $@ diff $@ $(<:.cpp=.output) rm $(<:.cpp=) $@ diff --git a/docs/docset/Info.plist b/docs/docset/Info.plist index 772ec08af..8f63bc5dd 100644 --- a/docs/docset/Info.plist +++ b/docs/docset/Info.plist @@ -13,7 +13,7 @@ dashIndexFilePath index.html DashDocSetFallbackURL - https://nlohmann.github.io/json/ + https://json.nlohmann.me/ isJavaScriptEnabled diff --git a/docs/docset/Makefile b/docs/docset/Makefile index 9f4363462..b638390db 100644 --- a/docs/docset/Makefile +++ b/docs/docset/Makefile @@ -52,35 +52,25 @@ install_docset_zeal: JSON_for_Modern_C++.docset mkdir -p $$docset_root; \ cp -r JSON_for_Modern_C++.docset $$docset_root/ +# both targets below compare the docset search index with the mkdocs page +# set. They share the same normalization (docs/foo/index.md and +# docs/foo.md both become foo/index.html, the URL mkdocs itself would +# give the page; the top-level index.md is excluded, as it is not part +# of the hand-curated docSet.sql) and use comm(1) on two sorted lists +# instead of running a sqlite3 query, or an O(n*m) nested shell loop, +# once per page. +DOCSET_INDEX_PATHS=$(shell sqlite3 docSet.dsidx "SELECT DISTINCT path FROM searchIndex" | sort) +DOCSET_PAGE_PATHS=$(shell echo '$(MKDOCS_PAGES)' | tr ' ' '\n' | grep -v '^index\.md$$' | $(SED) -E 's@/index\.md$$@/index.html@; s@\.md$$@/index.html@' | sort) + # list mkdocs pages missing from the docset index .PHONY: list_missing_pages list_missing_pages: docSet.dsidx - @for page in $(MKDOCS_PAGES); do \ - case "$$page" in \ - */index.md) path=$${page/\/index.md/} ;; \ - *) path=$${page/.md/} ;; \ - esac; \ - if [ "x$$page" != "xindex.md" -a "x$$(sqlite3 docSet.dsidx "SELECT COUNT(*) FROM searchIndex WHERE path='$$path/index.html'")" = "x0" ]; then \ - echo $$page; \ - fi \ - done + @comm -23 <(echo '$(DOCSET_PAGE_PATHS)' | tr ' ' '\n') <(echo '$(DOCSET_INDEX_PATHS)' | tr ' ' '\n') # list paths in the docset index without a corresponding mkdocs page .PHONY: list_removed_paths list_removed_paths: docSet.dsidx - @for path in $$(sqlite3 docSet.dsidx "SELECT path FROM searchIndex"); do \ - page=$${path/\/index.html/.md}; \ - page_index=$${path/index.html/index.md}; \ - page_found=0; \ - for p in $(MKDOCS_PAGES); do \ - if [ "x$$p" = "x$$page" -o "x$$p" = "x$$page_index" ]; then \ - page_found=1; \ - fi \ - done; \ - if [ "x$$page_found" = "x0" ]; then \ - echo $$path; \ - fi \ - done + @comm -13 <(echo '$(DOCSET_PAGE_PATHS)' | tr ' ' '\n') <(echo '$(DOCSET_INDEX_PATHS)' | tr ' ' '\n') .PHONY: clean clean: diff --git a/docs/docset/README.md b/docs/docset/README.md index 79a778eb8..9d962a91c 100644 --- a/docs/docset/README.md +++ b/docs/docset/README.md @@ -7,10 +7,11 @@ documentation browsers like [Dash](https://kapeli.com/dash), [Velocity](https:// The docset can be created with ```sh -make nlohmann_json.docset +make JSON_for_Modern_C++.docset ``` -The generated folder `nlohmann_json.docset` can then be opened in the documentation browser. +The generated folder `JSON_for_Modern_C++.docset` can then be opened in the documentation browser. `make all` builds a +`JSON_for_Modern_C++.tgz` archive instead, and `make install_docset_zeal` installs the docset for Zeal directly. A recent version is also part of the [Dash user contributions](https://github.com/Kapeli/Dash-User-Contributions/tree/master/docsets/JSON_for_Modern_C%2B%2B). diff --git a/docs/mkdocs/docs/api/basic_json/number_integer_t.md b/docs/mkdocs/docs/api/basic_json/number_integer_t.md index 9a2ffab7f..9b1d7ae74 100644 --- a/docs/mkdocs/docs/api/basic_json/number_integer_t.md +++ b/docs/mkdocs/docs/api/basic_json/number_integer_t.md @@ -51,7 +51,7 @@ range will yield over/underflow when used in a constructor. During deserializati will automatically be stored as [`number_unsigned_t`](number_unsigned_t.md) or [`number_float_t`](number_float_t.md). [RFC 8259](https://tools.ietf.org/html/rfc8259) further states: -> Note that when such software is used, numbers that are integers and are in the range $[-2^{53}+1, 2^{53}-1]$ are +> Note that when such software is used, numbers that are integers and are in the range [-253+1, 253-1] are > interoperable in the sense that implementations will agree exactly on their numeric values. As this range is a subrange of the exactly supported range [INT64_MIN, INT64_MAX], this class's integer type is diff --git a/docs/mkdocs/docs/api/basic_json/number_unsigned_t.md b/docs/mkdocs/docs/api/basic_json/number_unsigned_t.md index 674f7711d..f4799b2e9 100644 --- a/docs/mkdocs/docs/api/basic_json/number_unsigned_t.md +++ b/docs/mkdocs/docs/api/basic_json/number_unsigned_t.md @@ -52,7 +52,7 @@ when used in a constructor. During deserialization, too large or small integer n as [`number_integer_t`](number_integer_t.md) or [`number_float_t`](number_float_t.md). [RFC 8259](https://tools.ietf.org/html/rfc8259) further states: -> Note that when such software is used, numbers that are integers and are in the range $[-2^{53}+1, 2^{53}-1]$ are +> Note that when such software is used, numbers that are integers and are in the range [-253+1, 253-1] are > interoperable in the sense that implementations will agree exactly on their numeric values. As this range is a subrange (when considered in conjunction with the `number_integer_t` type) of the exactly supported diff --git a/docs/mkdocs/docs/api/basic_json/parse_error.md b/docs/mkdocs/docs/api/basic_json/parse_error.md index 74e16f41a..c3f49ef13 100644 --- a/docs/mkdocs/docs/api/basic_json/parse_error.md +++ b/docs/mkdocs/docs/api/basic_json/parse_error.md @@ -54,7 +54,7 @@ classDiagram ## Notes -For an input with $n$ bytes, 1 is the index of the first character and $n+1$ is the index of the terminating null byte +For an input with n bytes, 1 is the index of the first character and n+1 is the index of the terminating null byte or the end of file. This also holds true when reading a byte vector for binary formats. ## Examples diff --git a/docs/mkdocs/docs/examples/parse__iterator_pair.link b/docs/mkdocs/docs/examples/parse__iterator_pair.link deleted file mode 100644 index f464e54c8..000000000 --- a/docs/mkdocs/docs/examples/parse__iterator_pair.link +++ /dev/null @@ -1 +0,0 @@ -online \ No newline at end of file diff --git a/docs/mkdocs/docs/examples/parse__pointers.link b/docs/mkdocs/docs/examples/parse__pointers.link deleted file mode 100644 index 9a93ef1c5..000000000 --- a/docs/mkdocs/docs/examples/parse__pointers.link +++ /dev/null @@ -1 +0,0 @@ -online \ No newline at end of file diff --git a/docs/mkdocs/docs/features/binary_formats/bjdata.md b/docs/mkdocs/docs/features/binary_formats/bjdata.md index a0c84edaf..cd73b07e2 100644 --- a/docs/mkdocs/docs/features/binary_formats/bjdata.md +++ b/docs/mkdocs/docs/features/binary_formats/bjdata.md @@ -61,7 +61,7 @@ The library uses the following mapping from JSON values types to BJData types ac The following values can **not** be converted to a BJData value: - - strings with more than 18446744073709551615 bytes, i.e., $2^{64}-1$ bytes (theoretical) + - strings with more than 18446744073709551615 bytes, i.e., 264-1 bytes (theoretical) !!! info "Unused BJData markers" diff --git a/docs/mkdocs/docs/features/types/index.md b/docs/mkdocs/docs/features/types/index.md index e6078b825..354acd5ac 100644 --- a/docs/mkdocs/docs/features/types/index.md +++ b/docs/mkdocs/docs/features/types/index.md @@ -273,7 +273,7 @@ When the default type is used, the maximal unsigned integer number that can be s [RFC 8259](https://tools.ietf.org/html/rfc8259) further states: -> Note that when such software is used, numbers that are integers and are in the range $[-2^{53}+1, 2^{53}-1]$ are interoperable in the sense that implementations will agree exactly on their numeric values. +> Note that when such software is used, numbers that are integers and are in the range [-253+1, 253-1] are interoperable in the sense that implementations will agree exactly on their numeric values. As this range is a subrange of the exactly supported range [`INT64_MIN`, `INT64_MAX`], this class's integer type is interoperable. diff --git a/docs/mkdocs/docs/features/types/number_handling.md b/docs/mkdocs/docs/features/types/number_handling.md index cf37b044a..ef43054e5 100644 --- a/docs/mkdocs/docs/features/types/number_handling.md +++ b/docs/mkdocs/docs/features/types/number_handling.md @@ -48,7 +48,7 @@ On number interoperability, the following remarks are made: for numeric magnitude and precision than is widely available. Note that when such software is used, numbers that are integers and - are in the range $[-2^{53}+1, 2^{53}-1]$ are interoperable in the + are in the range [-253+1, 253-1] are interoperable in the sense that implementations will agree exactly on their numeric values. @@ -95,9 +95,9 @@ This is the same behavior as the code `#!c double x = 3.141592653589793238462643 !!! success "Interoperability" - - The library is interoperable with respect to the specification, because its supported range $[-2^{63}, 2^{64}-1]$ is - larger than the described range $[-2^{53}+1, 2^{53}-1]$. - - All integers outside the range $[-2^{63}, 2^{64}-1]$, as well as floating-point numbers are stored as `double`. + - The library is interoperable with respect to the specification, because its supported range [-263, 264-1] is + larger than the described range [-253+1, 253-1]. + - All integers outside the range [-263, 264-1], as well as floating-point numbers are stored as `double`. This also concurs with the specification above. ### Zeros diff --git a/docs/mkdocs/docs/integration/package_managers.md b/docs/mkdocs/docs/integration/package_managers.md index 792a0fa5a..28f95e584 100644 --- a/docs/mkdocs/docs/integration/package_managers.md +++ b/docs/mkdocs/docs/integration/package_managers.md @@ -678,11 +678,11 @@ to install the [nlohmann-json](https://ports.macports.org/port/nlohmann-json/) p 1. Create the following files: ```cpp title="example.cpp" - --8<-- "integration/homebrew/example.cpp" + --8<-- "integration/macports/example.cpp" ``` ```cmake title="CMakeLists.txt" - --8<-- "integration/homebrew/CMakeLists.txt" + --8<-- "integration/macports/CMakeLists.txt" ``` 2. Install the package: diff --git a/docs/mkdocs/mkdocs.yml b/docs/mkdocs/mkdocs.yml index b31fb1734..4bd207e60 100644 --- a/docs/mkdocs/mkdocs.yml +++ b/docs/mkdocs/mkdocs.yml @@ -351,7 +351,6 @@ markdown_extensions: - toc: permalink: true - md_in_html - - pymdownx.arithmatex - pymdownx.betterem: smart_enable: all - pymdownx.caret @@ -443,6 +442,3 @@ plugins: extra_css: - css/custom.css - -extra_javascript: - - https://cdnjs.cloudflare.com/ajax/libs/mathjax/2.7.0/MathJax.js?config=TeX-MML-AM_CHTML diff --git a/docs/mkdocs/scripts/check_structure.py b/docs/mkdocs/scripts/check_structure.py index c8d637d06..78ad0d9fa 100755 --- a/docs/mkdocs/scripts/check_structure.py +++ b/docs/mkdocs/scripts/check_structure.py @@ -79,9 +79,9 @@ def check_structure() -> None: report("whitespace/line_length", f"{file}:{lineno+1} ({current_section})", f"line is too long ({len(line)} vs. 160 chars)") # sections in `` comments are treated as present - if line.startswith("") + nolint_match = re.match(r"", line) + if nolint_match: + current_section = nolint_match.group(1) existing_sections.append(current_section) # check if sections are correct @@ -97,7 +97,7 @@ def check_structure() -> None: if len(unexpected): report("style/numbering", f"{file}:{lineno} ({current_section})", f'unexpected overloads: {", ".join([f"({x})" for x in unexpected])}') - current_section = line.strip("## ") + current_section = line[3:] existing_sections.append(current_section) if current_section in expected_sections: @@ -141,7 +141,7 @@ def check_structure() -> None: # check that non-example admonitions have titles untitled_admonition = re.match(r"^(\?\?\?|!!!) ([^ ]+)$", line) if untitled_admonition and untitled_admonition.group(2) != "example": - report("style/admonition_title", f"{file}:{lineno} ({current_section})", f'"{untitled_admonition.group(2)}" admonitions should have a title') + report("style/admonition_title", f"{file}:{lineno+1} ({current_section})", f'"{untitled_admonition.group(2)}" admonitions should have a title') previous_line = line From 35802e78d6b28d7973ac3235988d834468da1539 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Wed, 30 Sep 2026 22:21:27 +0200 Subject: [PATCH 6/7] Remove dead Makefile targets; document macro_builder; tidy serve_header (#5735) * Remove dead doctest help entry and pretty_format target from Makefile The top-level Makefile still carried three leftovers: - The help text listed a "doctest" target that was removed in #4560, so "make doctest" fails with "No rule to make target". The example check now runs as "make check_output -C docs". - "pretty_format" ran clang-format on all sources, but .clang-format was deleted in #4573, so the target reformatted everything in the default LLVM style, against the Artistic Style formatting that "make pretty" applies and CI enforces. - "clean" removed benchmarks/files/numbers/*.json, a directory that no longer exists since the benchmarks moved to tests/benchmarks (#3462). Only maintainer tooling changes; the library is not affected. Part of #5717 Signed-off-by: Niels Lohmann * Document tools/macro_builder and tidy up serve_header.py tools/macro_builder generates the NLOHMANN_JSON_EXPAND, NLOHMANN_JSON_GET_MACRO and NLOHMANN_JSON_PASTE* macros in macro_scope.hpp, but nothing referred to it. Add a README that explains what it generates, how to run it and where the output goes, and which dependent tables (NLOHMANN_JSON_DOUBLE_PASTE, NLOHMANN_JSON_TYPE_BODY) are maintained by hand. Point to it from a comment above NLOHMANN_JSON_EXPAND. The generator itself is unchanged; following the README reproduces the header byte for byte. In serve_header.py, drop the LGTM suppression (LGTM.com shut down in 2022), replace the """.""" placeholder docstrings with real ones, and import socket and ssl at module level. DualStackServer.server_bind uses socket, which was only imported under __main__; when the module was imported instead, the NameError was swallowed and IPV6_V6ONLY was not cleared. The header change is a comment only; behavior, API and ABI are unchanged. Part of #5717 Signed-off-by: Niels Lohmann * Hash every release_files artifact, not a hardcoded subset The `release` target signed and copied json_fwd.hpp into release_files alongside json.hpp, but the shasum line that writes hashes.txt only listed json.hpp, include.zip and json.tar.xz. Users could not verify the published json_fwd.hpp against hashes.txt. Hash every file in release_files except the .asc signatures instead of naming files by hand, so a newly shipped header (such as the json_literals.hpp that #5610 adds to this target) cannot be missed again. Only affects the generated hashes.txt release artifact; the library itself is unaffected. Overlaps #5610, which touches the same lines to add json_literals.hpp to the release target. #5717 item 1 Signed-off-by: Niels Lohmann * Remove the broken fuzz_testing* Makefile targets fuzz_testing and fuzz_testing_{bon8,bson,cbor,msgpack,ubjson} seeded fuzz-testing/testcases from tests/data, which was removed in dbf1a1f41 (2020) when the test data moved to the external json_test_data repo. The find command found nothing, but the pipeline's exit status was that of xargs, so the recipe still reported success with an empty corpus, and the printed afl-fuzz command would refuse to start. The recipes were also six near-identical copies with unquoted -name patterns, used the legacy CXX=afl-clang++, and fuzzing-start/stop were missing from both the help output and .PHONY. tests/fuzzing.md already documents the working flow (download json_test_data, then `make -C tests fuzzers`), so replace the six broken targets and their help lines with a single pointer to that document instead of trying to keep six copies of a fragile shell pipeline in sync. This does not affect OSS-Fuzz, which builds through tests/Makefile. Overlaps #5621, which adds a seventh copy of the same broken line for fuzz_testing_json_view. #5717 item 2 Signed-off-by: Niels Lohmann * Remove stale Travis comment above check-amalgamation check-amalgamation carried "Note: this target is called by Travis", left over from before the project switched off Travis CI. The prior Makefile cleanup commit removed the other stale Travis-era leftovers (the doctest help entry, pretty_format, and the benchmarks/ path in clean) but missed this comment. #5717 item 4 Signed-off-by: Niels Lohmann * Fix stale install/usage instructions in the vendored amalgamate README tools/amalgamate/README.md is the unmodified upstream text and no longer matches how the tool is used here: - It named a Bitbucket origin that no longer exists; CHANGES.md already tracks the GitHub mirror commit this copy is based on. - It asked for Python 2.7, but CI and the Makefile run the script with python3. - It told readers to run ./test.sh (not vendored) and install to /usr/local/bin; in this repository the tool runs through `make amalgamate`. - Its usage synopsis showed `-v` taking no argument, but the script's own argparser requires `choices=["yes", "no"]`, so that form fails with "argument -v/--verbose: expected one argument". The Makefile calls it as `--verbose=yes`. - It pointed at test/source.c.json and test/include.h.json, which are not vendored; the configs actually used are config_json.json and config_json_fwd.json. Rewrote only the Installing and Using sections to match; left the "Here be dragons" caveats and the rest of the vendored code untouched to avoid diverging further from upstream. Overlaps #5615, which edits amalgamate.py, this README and CHANGES.md. #5717 item 6 Signed-off-by: Niels Lohmann * Derive generate_natvis.py's ABI tag list and version from abi_macros.hpp generate_natvis.py hard-coded abi_tags = ['_diag', '_ldvcmp', '_dp', '_bics', '_psp', '_snul'] and required --version on the command line. The source of truth is include/nlohmann/detail/abi_macros.hpp: the NLOHMANN_JSON_ABI_TAG_* defines, the argument order of NLOHMANN_JSON_ABI_TAGS_CONCAT, and NLOHMANN_JSON_VERSION_MAJOR/MINOR/ PATCH. Nothing checked that the copies stayed in sync, and they have drifted apart before: _dp was added in #4517 but missed here until #5544, and 3.11.3 shipped json_abi_v3_11_2 namespaces (#4340). Parse the tag list (in NLOHMANN_JSON_ABI_TAGS_CONCAT order) and the version from abi_macros.hpp instead of hard-coding them. Make --version optional (falling back to the parsed version) and default the output directory to the repository root the script lives in. Add a "natvis" Makefile target that runs the script, and extend check-amalgamation to regenerate nlohmann_json.natvis and fail on a diff, the same way it already does for the amalgamated headers and BUILD.bazel. Wire the same regeneration into check_amalgamation.yml, using the tool copy checked out from develop (as the workflow already does for amalgamate.py) and installing jinja2 from tools/generate_natvis/requirements.txt. Update the tool's README to say it must be re-run after adding an ABI tag or bumping the version. Verified: a run against develop produces no diff (with either the default or an explicit --version 3.12.0); adding a dummy NLOHMANN_JSON_ABI_TAG_* without a matching #define makes the script fail loudly instead of silently omitting the tag; xmllint --noout passes on the regenerated file; and running the script from a directory other than the one being checked (simulating the workflow's separate tool checkout) against this repository root also produces no diff. Overlaps #5600, which added _ekmo to the same hand-written abi_tags line and regenerated the file. #5717 item 3 Signed-off-by: Niels Lohmann * Strip the leading "./" find(1) prefix from release hashes.txt entries bc6e7db72 (#5717 item 1) switched the release target's shasum line from naming files by hand to $$(find . -type f -not -name '*.asc' | sort), so a newly shipped header is hashed automatically. Run from inside release_files, that find prints paths as "./json.hpp" instead of "json.hpp", so hashes.txt lists "./json.hpp" etc. instead of the plain filenames it always used. shasum -c still verifies "./json.hpp" fine, but it is a needless cosmetic regression for anyone reading the file or matching it against release notes. Strip the "./" prefix with sed before sorting, keeping the filenames exactly as before while still hashing every artifact automatically. Review fix for #5717 item 1 (PR #5735). Signed-off-by: Niels Lohmann * Fix check_amalgamation.yml: pass --version to generate_natvis.py bff45f111 (#5717 item 3) made --version optional in tools/generate_natvis/generate_natvis.py and wired the workflow's new "Regenerate nlohmann_json.natvis" step to call it without --version, relying on the script deriving the version from abi_macros.hpp itself. But NATVIS_TOOL_DIR is checked out from develop, the same way TOOL_DIR already is for amalgamate.py, precisely so an in-flight PR's tooling changes cannot mark themselves clean. Until this PR (or an equivalent) merges to develop, that checkout is the old generate_natvis.py, whose --version argument is still required=True. The new step's invocation of "generate_natvis.py $MAIN_DIR" (no --version) then fails argparse on this PR's own CI run with "the following arguments are required: --version", before the check ever gets to compare output. Extract the version from $MAIN_DIR's own abi_macros.hpp in the workflow and always pass it as --version. That satisfies the old script's required argument and is accepted as an explicit override by the new one, so the step behaves the same whether NATVIS_TOOL_DIR holds the pre- or post-merge tool, and stays correct for later PRs that bump the version. Verified by running the workflow step's shell logic locally against both the pre-#5717 generate_natvis.py (checked out at 633de8e44) and the new one: both produce the identical nlohmann_json.natvis as the committed file. Review fix for #5717 item 3 (PR #5735). Signed-off-by: Niels Lohmann * Extend tools/macro_builder to also generate DOUBLE_PASTE and TYPE_BODY tools/macro_builder only emitted NLOHMANN_JSON_EXPAND..PASTE64, so two other tables that scale with the same max_args stayed hand-maintained with nothing checking them: NLOHMANN_JSON_DOUBLE_PASTE (added by hand in #4563 for the *_WITH_NAMES macros) and the 64-slot dispatch table of NLOHMANN_JSON_TYPE_BODY (the #4041 zero-member/one-or-more-member switch). Both tables pass one macro name per slot to the same NLOHMANN_JSON_GET_MACRO dispatch as PASTE, so they can drift out of sync with max_args exactly the way _dp did in the ABI tag list fixed by #5544. Extend main.cpp with build_double_paste_code() (same recursive-doubling shape as build_paste_code(), but DOUBLE_PASTE consumes two arguments per member, so an even slot index falls back to the next lower odd DOUBLE_PASTE) and build_type_body_table() (max_args - 1 MEMBERS slots and one trailing EMPTY slot, 8 per line, matching how it is written by hand today). Add a "type_body" argument that selects the TYPE_BODY block, since it lives at a separate location in macro_scope.hpp from the EXPAND..DOUBLE_PASTE63 block; plain invocation is unchanged apart from covering the extended range. No longer emit the tool's old trailing blank line, so its output is directly diffable without post-processing. Verified with c++ -std=c++11: running the tool (with and without "type_body") and piping the raw output through the pinned astyle reproduces both blocks of the current macro_scope.hpp byte for byte. tests/src/unit-udt_macro.cpp (all NLOHMANN_DEFINE_TYPE_*/_WITH_NAMES/ zero-member variants) passes unchanged under -std=c++11 and -std=c++17 with -fsanitize=address,undefined. Add a "macro_builder_check" Makefile target that builds main.cpp, regenerates both blocks into a scratch directory inside the repository (astyle's --project lookup needs the target files under the same tree as .astylerc, unlike an external /tmp directory), and diffs them against the corresponding ranges of macro_scope.hpp; wire it into check-amalgamation next to the natvis check. Wire the same regeneration into check_amalgamation.yml, splicing the (still unindented) generated blocks back into the PR's own macro_scope.hpp before the existing astyle/amalgamation step runs, so that step's own tree-wide astyle pass both indents them and folds any drift into the amalgamation patch/diff the workflow already produces. Unlike amalgamate.py and generate_natvis.py, this step builds tools/macro_builder/main.cpp from the pull request's own checkout ($MAIN_DIR) rather than a separate checkout of tools/ at develop: this tool has no independent source of truth to regenerate against (its README documents that it must reproduce macro_scope.hpp byte for byte), so a develop-pinned copy would only reproduce the generate_natvis.py trap fixed in a previous commit on this branch, where a PR that teaches the tool to cover more of the file fails its own CI until that PR merges and updates the develop copy. Add tools/macro_builder/README.md documentation for both new tables and the two-invocation usage, and a short pointer comment above NLOHMANN_JSON_TYPE_BODY (the EXPAND pointer already covered the first block; extended its wording to include DOUBLE_PASTE63). Closes #5717 item 5 in full, completing what the documentation-only "Document tools/macro_builder..." commit already on this branch left open (that commit's README/pointer-comment half stands; it also covers item 7). Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- .github/workflows/check_amalgamation.yml | 56 +++++++++++ Makefile | 115 +++++++---------------- include/nlohmann/detail/macro_scope.hpp | 5 + single_include/nlohmann/json.hpp | 5 + tools/amalgamate/README.md | 26 ++--- tools/generate_natvis/README.md | 20 +++- tools/generate_natvis/generate_natvis.py | 62 +++++++++++- tools/macro_builder/README.md | 51 ++++++++++ tools/macro_builder/main.cpp | 87 ++++++++++++++--- tools/serve_header/serve_header.py | 22 ++--- 10 files changed, 328 insertions(+), 121 deletions(-) create mode 100644 tools/macro_builder/README.md diff --git a/.github/workflows/check_amalgamation.yml b/.github/workflows/check_amalgamation.yml index 35f3a57f7..d81a80b18 100644 --- a/.github/workflows/check_amalgamation.yml +++ b/.github/workflows/check_amalgamation.yml @@ -35,6 +35,7 @@ jobs: MAIN_DIR: ${{ github.workspace }}/main INCLUDE_DIR: ${{ github.workspace }}/main/single_include/nlohmann TOOL_DIR: ${{ github.workspace }}/tools/tools/amalgamate + NATVIS_TOOL_DIR: ${{ github.workspace }}/tools/tools/generate_natvis steps: - name: Harden Runner @@ -61,6 +62,48 @@ jobs: python3 -mvenv venv venv/bin/pip3 install -r $MAIN_DIR/tools/astyle/requirements.txt + - name: Install generate_natvis dependencies + run: pip3 install -r $NATVIS_TOOL_DIR/requirements.txt + + - name: Regenerate the tools/macro_builder tables in macro_scope.hpp + run: | + cd $MAIN_DIR + + # Built from this PR's own tools/macro_builder/main.cpp, not a + # develop checkout: unlike amalgamate.py and generate_natvis.py, + # this tool has no other source of truth to check against (its + # own README documents that it must reproduce macro_scope.hpp + # byte for byte), so there is nothing to gain from checking out a + # separate copy, and doing so would make this step fail on a PR + # that adds support for a new dispatch table until that PR itself + # merges to develop, the same way generate_natvis.py's --version + # requirement briefly did. + TMPDIR=$(mktemp -d ./macro_builder_check.XXXXXX) + c++ -std=c++11 tools/macro_builder/main.cpp -o "$TMPDIR/macro_builder" + "$TMPDIR/macro_builder" > "$TMPDIR/paste.hpp" + "$TMPDIR/macro_builder" type_body > "$TMPDIR/type_body.hpp" + + # Splice the (still unindented) generated blocks back into + # macro_scope.hpp; the astyle pass below indents their + # continuation lines the same way it does for the rest of + # include/, so a correctly regenerated file comes out unchanged. + awk -v newfile="$TMPDIR/paste.hpp" ' + BEGIN { while ((getline line < newfile) > 0) { new = new line "\n" } } + /^#define NLOHMANN_JSON_EXPAND\( x \) x$/ { printf "%s", new; skip=1 } + skip && /^#define NLOHMANN_JSON_DOUBLE_PASTE63\(/ { skip=0; next } + skip { next } + { print } + ' include/nlohmann/detail/macro_scope.hpp > "$TMPDIR/macro_scope_1.hpp" + awk -v newfile="$TMPDIR/type_body.hpp" ' + BEGIN { while ((getline line < newfile) > 0) { new = new line "\n" } } + /^#define NLOHMANN_JSON_TYPE_BODY\(Prefix, \.\.\.\)/ { printf "%s", new; skip=1 } + skip && /^[[:space:]]*NLOHMANN_JSON_TYPE_BODY_SENTINEL\)\)$/ { skip=0; next } + skip { next } + { print } + ' "$TMPDIR/macro_scope_1.hpp" > "$TMPDIR/macro_scope_2.hpp" + mv "$TMPDIR/macro_scope_2.hpp" include/nlohmann/detail/macro_scope.hpp + rm -rf "$TMPDIR" + - name: Regenerate amalgamation, formatting, and BUILD.bazel run: | cd $MAIN_DIR @@ -88,6 +131,19 @@ jobs: ${{ github.workspace }}/venv/bin/astyle --project=tools/astyle/.astylerc --suffix=none --quiet \ $(find $SOURCE_DIRS -type f \( -name '*.hpp' -o -name '*.cpp' -o -name '*.cu' \) -not -path 'tests/thirdparty/*' -not -path 'tests/abi/include/nlohmann/*' | sort) + - name: Regenerate nlohmann_json.natvis + run: | + cd $MAIN_DIR + # Pass --version explicitly so this step also works with the tool + # copy from develop before this repository's own generate_natvis.py + # learns to derive the version itself: the older script requires + # --version, and the newer one accepts it as an explicit override. + ABI_MACROS=include/nlohmann/detail/abi_macros.hpp + VERSION_MAJOR=$(grep -m1 'define NLOHMANN_JSON_VERSION_MAJOR' $ABI_MACROS | grep -o '[0-9]\+') + VERSION_MINOR=$(grep -m1 'define NLOHMANN_JSON_VERSION_MINOR' $ABI_MACROS | grep -o '[0-9]\+') + VERSION_PATCH=$(grep -m1 'define NLOHMANN_JSON_VERSION_PATCH' $ABI_MACROS | grep -o '[0-9]\+') + python3 $NATVIS_TOOL_DIR/generate_natvis.py --version "$VERSION_MAJOR.$VERSION_MINOR.$VERSION_PATCH" $MAIN_DIR + - name: Build patch and check for differences id: diff run: | diff --git a/Makefile b/Makefile index ccea759af..fec936343 100644 --- a/Makefile +++ b/Makefile @@ -1,4 +1,4 @@ -.PHONY: pretty clean ChangeLog.md release update_hedley update_hedley_undef BUILD.bazel +.PHONY: pretty clean ChangeLog.md release update_hedley update_hedley_undef BUILD.bazel natvis macro_builder_check ########################################################################## # configuration @@ -24,6 +24,9 @@ AMALGAMATED_FWD_FILE=single_include/nlohmann/json_fwd.hpp # json_literals.hpp only includes , so it is copied verbatim AMALGAMATED_LITERALS_FILE=single_include/nlohmann/json_literals.hpp +# the header with the argument-counting macros generated by tools/macro_builder +MACRO_SCOPE_HPP=include/nlohmann/detail/macro_scope.hpp + ########################################################################## # documentation of the Makefile's targets @@ -36,13 +39,9 @@ all: @echo "ChangeLog.md - generate ChangeLog file" @echo "check-amalgamation - check whether sources have been amalgamated and BUILD.bazel is up to date" @echo "clean - remove built files" - @echo "doctest - compile example files and check their output" - @echo "fuzz_testing - prepare fuzz testing of the JSON parser" - @echo "fuzz_testing_bon8 - prepare fuzz testing of the BON8 parser" - @echo "fuzz_testing_bson - prepare fuzz testing of the BSON parser" - @echo "fuzz_testing_cbor - prepare fuzz testing of the CBOR parser" - @echo "fuzz_testing_msgpack - prepare fuzz testing of the MessagePack parser" - @echo "fuzz_testing_ubjson - prepare fuzz testing of the UBJSON parser" + @echo "fuzzing - see tests/fuzzing.md for how to build and run the fuzzers" + @echo "macro_builder_check - check that macro_scope.hpp matches tools/macro_builder's output" + @echo "natvis - regenerate nlohmann_json.natvis from the current ABI tags and version" @echo "pretty - beautify code with Artistic Style" @echo "run_benchmarks - build and run benchmarks" @echo "update_hedley - download Hedley and regenerate hedley.hpp / hedley_undef.hpp" @@ -61,74 +60,6 @@ run_benchmarks: cd cmake-build-benchmarks ; ./json_benchmarks -########################################################################## -# fuzzing -########################################################################## - -# the overall fuzz testing target -fuzz_testing: - rm -fr fuzz-testing - mkdir -p fuzz-testing fuzz-testing/testcases fuzz-testing/out - $(MAKE) parse_afl_fuzzer -C tests CXX=afl-clang++ - mv tests/parse_afl_fuzzer fuzz-testing/fuzzer - find tests/data/json_tests -size -5k -name *json | xargs -I{} cp "{}" fuzz-testing/testcases - @echo "Execute: afl-fuzz -i fuzz-testing/testcases -o fuzz-testing/out fuzz-testing/fuzzer" - -fuzz_testing_bon8: - rm -fr fuzz-testing - mkdir -p fuzz-testing fuzz-testing/testcases fuzz-testing/out - $(MAKE) parse_bon8_fuzzer -C tests CXX=afl-clang++ - mv tests/parse_bon8_fuzzer fuzz-testing/fuzzer - find tests/data -size -5k -name *.bon8 | xargs -I{} cp "{}" fuzz-testing/testcases - @echo "Execute: afl-fuzz -i fuzz-testing/testcases -o fuzz-testing/out fuzz-testing/fuzzer" - -fuzz_testing_bson: - rm -fr fuzz-testing - mkdir -p fuzz-testing fuzz-testing/testcases fuzz-testing/out - $(MAKE) parse_bson_fuzzer -C tests CXX=afl-clang++ - mv tests/parse_bson_fuzzer fuzz-testing/fuzzer - find tests/data -size -5k -name *.bson | xargs -I{} cp "{}" fuzz-testing/testcases - @echo "Execute: afl-fuzz -i fuzz-testing/testcases -o fuzz-testing/out fuzz-testing/fuzzer" - -fuzz_testing_cbor: - rm -fr fuzz-testing - mkdir -p fuzz-testing fuzz-testing/testcases fuzz-testing/out - $(MAKE) parse_cbor_fuzzer -C tests CXX=afl-clang++ - mv tests/parse_cbor_fuzzer fuzz-testing/fuzzer - find tests/data -size -5k -name *.cbor | xargs -I{} cp "{}" fuzz-testing/testcases - @echo "Execute: afl-fuzz -i fuzz-testing/testcases -o fuzz-testing/out fuzz-testing/fuzzer" - -fuzz_testing_msgpack: - rm -fr fuzz-testing - mkdir -p fuzz-testing fuzz-testing/testcases fuzz-testing/out - $(MAKE) parse_msgpack_fuzzer -C tests CXX=afl-clang++ - mv tests/parse_msgpack_fuzzer fuzz-testing/fuzzer - find tests/data -size -5k -name *.msgpack | xargs -I{} cp "{}" fuzz-testing/testcases - @echo "Execute: afl-fuzz -i fuzz-testing/testcases -o fuzz-testing/out fuzz-testing/fuzzer" - -fuzz_testing_ubjson: - rm -fr fuzz-testing - mkdir -p fuzz-testing fuzz-testing/testcases fuzz-testing/out - $(MAKE) parse_ubjson_fuzzer -C tests CXX=afl-clang++ - mv tests/parse_ubjson_fuzzer fuzz-testing/fuzzer - find tests/data -size -5k -name *.ubjson | xargs -I{} cp "{}" fuzz-testing/testcases - @echo "Execute: afl-fuzz -i fuzz-testing/testcases -o fuzz-testing/out fuzz-testing/fuzzer" - -fuzzing-start: - afl-fuzz -S fuzzer1 -i fuzz-testing/testcases -o fuzz-testing/out fuzz-testing/fuzzer > /dev/null & - afl-fuzz -S fuzzer2 -i fuzz-testing/testcases -o fuzz-testing/out fuzz-testing/fuzzer > /dev/null & - afl-fuzz -S fuzzer3 -i fuzz-testing/testcases -o fuzz-testing/out fuzz-testing/fuzzer > /dev/null & - afl-fuzz -S fuzzer4 -i fuzz-testing/testcases -o fuzz-testing/out fuzz-testing/fuzzer > /dev/null & - afl-fuzz -S fuzzer5 -i fuzz-testing/testcases -o fuzz-testing/out fuzz-testing/fuzzer > /dev/null & - afl-fuzz -S fuzzer6 -i fuzz-testing/testcases -o fuzz-testing/out fuzz-testing/fuzzer > /dev/null & - afl-fuzz -S fuzzer7 -i fuzz-testing/testcases -o fuzz-testing/out fuzz-testing/fuzzer > /dev/null & - afl-fuzz -M fuzzer0 -i fuzz-testing/testcases -o fuzz-testing/out fuzz-testing/fuzzer - -fuzzing-stop: - -killall fuzzer - -killall afl-fuzz - - ########################################################################## # Static analysis ########################################################################## @@ -158,10 +89,6 @@ install_astyle: pretty: install_astyle $(ASTYLE) --project=tools/astyle/.astylerc $(SRCS) $(TESTS_SRCS) $(AMALGAMATED_FILE) $(AMALGAMATED_FWD_FILE) $(AMALGAMATED_LITERALS_FILE) docs/mkdocs/docs/examples/*.cpp -# call the Clang-Format on all source files -pretty_format: - for FILE in $(SRCS) $(TESTS_SRCS) $(AMALGAMATED_FILE) docs/mkdocs/docs/examples/*.cpp; do echo $$FILE; clang-format -i $$FILE; done - # create single header files and pretty print amalgamate: $(AMALGAMATED_FILE) $(AMALGAMATED_FWD_FILE) $(AMALGAMATED_LITERALS_FILE) $(MAKE) pretty @@ -178,8 +105,26 @@ $(AMALGAMATED_FWD_FILE): $(SRCS) $(AMALGAMATED_LITERALS_FILE): include/nlohmann/json_literals.hpp cp include/nlohmann/json_literals.hpp $(AMALGAMATED_LITERALS_FILE) +# regenerate nlohmann_json.natvis from the ABI tags and version in include/nlohmann/detail/abi_macros.hpp +natvis: + python3 tools/generate_natvis/generate_natvis.py . + +# regenerate the two tools/macro_builder blocks of $(MACRO_SCOPE_HPP) (see its README.md) and diff against the +# checked-in header; phony, because it never writes $(MACRO_SCOPE_HPP) itself +macro_builder_check: + @set -e; \ + TMPDIR=$$(mktemp -d ./macro_builder_check.XXXXXX); \ + trap 'rm -rf "$$TMPDIR"' EXIT; \ + $(CXX) -std=c++11 tools/macro_builder/main.cpp -o "$$TMPDIR/macro_builder"; \ + "$$TMPDIR/macro_builder" > "$$TMPDIR/paste.hpp"; \ + "$$TMPDIR/macro_builder" type_body > "$$TMPDIR/type_body.hpp"; \ + $(ASTYLE) --project=tools/astyle/.astylerc --suffix=none --quiet "$$TMPDIR/paste.hpp" "$$TMPDIR/type_body.hpp"; \ + sed -n '/^#define NLOHMANN_JSON_EXPAND( x ) x$$/,/^#define NLOHMANN_JSON_DOUBLE_PASTE63(/p' $(MACRO_SCOPE_HPP) > "$$TMPDIR/paste_actual.hpp"; \ + sed -n '/^#define NLOHMANN_JSON_TYPE_BODY(Prefix, \.\.\.)/,/^ NLOHMANN_JSON_TYPE_BODY_SENTINEL))$$/p' $(MACRO_SCOPE_HPP) > "$$TMPDIR/type_body_actual.hpp"; \ + diff "$$TMPDIR/paste.hpp" "$$TMPDIR/paste_actual.hpp" || (echo "===================================================================\n $(MACRO_SCOPE_HPP) (NLOHMANN_JSON_EXPAND..NLOHMANN_JSON_DOUBLE_PASTE63) is out of date!\n Regenerate it, see tools/macro_builder/README.md.\n===================================================================" ; exit 1); \ + diff "$$TMPDIR/type_body.hpp" "$$TMPDIR/type_body_actual.hpp" || (echo "===================================================================\n $(MACRO_SCOPE_HPP) (NLOHMANN_JSON_TYPE_BODY) is out of date!\n Regenerate it, see tools/macro_builder/README.md.\n===================================================================" ; exit 1) + # check if file single_include/nlohmann/json.hpp has been amalgamated from the nlohmann sources -# Note: this target is called by Travis check-amalgamation: @mv $(AMALGAMATED_FILE) $(AMALGAMATED_FILE)~ @mv $(AMALGAMATED_FWD_FILE) $(AMALGAMATED_FWD_FILE)~ @@ -195,6 +140,11 @@ check-amalgamation: @$(MAKE) BUILD.bazel @diff BUILD.bazel BUILD.bazel~ || (echo "===================================================================\n BUILD.bazel is out of date! Please run 'make BUILD.bazel'.\n===================================================================" ; mv BUILD.bazel~ BUILD.bazel ; false) @mv BUILD.bazel~ BUILD.bazel + @mv nlohmann_json.natvis nlohmann_json.natvis~ + @$(MAKE) natvis + @diff nlohmann_json.natvis nlohmann_json.natvis~ || (echo "===================================================================\n nlohmann_json.natvis is out of date! Please run 'make natvis'.\n===================================================================" ; mv nlohmann_json.natvis~ nlohmann_json.natvis ; false) + @mv nlohmann_json.natvis~ nlohmann_json.natvis + @$(MAKE) macro_builder_check # generate the Bazel BUILD file; phony, because a removed header would not trigger a rebuild BUILD.bazel: @@ -246,7 +196,7 @@ release: include.zip json.tar.xz cp $(AMALGAMATED_FWD_FILE) release_files cp $(AMALGAMATED_LITERALS_FILE) release_files mv $(AMALGAMATED_FILE).asc $(AMALGAMATED_FWD_FILE).asc $(AMALGAMATED_LITERALS_FILE).asc json.tar.xz json.tar.xz.asc include.zip include.zip.asc release_files - cd release_files ; shasum -a 256 json.hpp include.zip json.tar.xz > hashes.txt + cd release_files ; shasum -a 256 $$(find . -type f -not -name '*.asc' | sed 's|^\./||' | sort) > hashes.txt ########################################################################## @@ -256,7 +206,6 @@ release: include.zip json.tar.xz # clean up clean: rm -fr fuzz fuzz-testing *.dSYM tests/*.dSYM - rm -fr benchmarks/files/numbers/*.json rm -fr cmake-build-benchmarks fuzz-testing cmake-build-pvs-studio release_files $(MAKE) clean -Cdocs diff --git a/include/nlohmann/detail/macro_scope.hpp b/include/nlohmann/detail/macro_scope.hpp index 2d0fccf18..7c5ae3089 100644 --- a/include/nlohmann/detail/macro_scope.hpp +++ b/include/nlohmann/detail/macro_scope.hpp @@ -359,6 +359,7 @@ void templated_json_throw(ExceptionType exception) // Macros to simplify conversion from/to types +// NLOHMANN_JSON_EXPAND to NLOHMANN_JSON_DOUBLE_PASTE63 are generated by tools/macro_builder (see its README.md) #define NLOHMANN_JSON_EXPAND( x ) x #define NLOHMANN_JSON_GET_MACRO(_1, _2, _3, _4, _5, _6, _7, _8, _9, _10, _11, _12, _13, _14, _15, _16, _17, _18, _19, _20, _21, _22, _23, _24, _25, _26, _27, _28, _29, _30, _31, _32, _33, _34, _35, _36, _37, _38, _39, _40, _41, _42, _43, _44, _45, _46, _47, _48, _49, _50, _51, _52, _53, _54, _55, _56, _57, _58, _59, _60, _61, _62, _63, _64, NAME,...) NAME #define NLOHMANN_JSON_PASTE(...) NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_GET_MACRO(__VA_ARGS__, \ @@ -621,6 +622,10 @@ void templated_json_throw(ExceptionType exception) // arguments, so dispatching on Type,BaseType,member... directly would run out // one slot early and cap the derived-type macros at 62 members instead of the // 63 that NLOHMANN_JSON_PASTE supports. +// +// The slot table below (down to the closing NLOHMANN_JSON_TYPE_BODY_SENTINEL)) +// is generated by tools/macro_builder (see its README.md; run with the +// "type_body" argument). #define NLOHMANN_JSON_TYPE_BODY(Prefix, ...) NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_GET_MACRO(__VA_ARGS__, \ Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, \ Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, \ diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 1fa63f344..3437b5408 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -2772,6 +2772,7 @@ void templated_json_throw(ExceptionType exception) // Macros to simplify conversion from/to types +// NLOHMANN_JSON_EXPAND to NLOHMANN_JSON_DOUBLE_PASTE63 are generated by tools/macro_builder (see its README.md) #define NLOHMANN_JSON_EXPAND( x ) x #define NLOHMANN_JSON_GET_MACRO(_1, _2, _3, _4, _5, _6, _7, _8, _9, _10, _11, _12, _13, _14, _15, _16, _17, _18, _19, _20, _21, _22, _23, _24, _25, _26, _27, _28, _29, _30, _31, _32, _33, _34, _35, _36, _37, _38, _39, _40, _41, _42, _43, _44, _45, _46, _47, _48, _49, _50, _51, _52, _53, _54, _55, _56, _57, _58, _59, _60, _61, _62, _63, _64, NAME,...) NAME #define NLOHMANN_JSON_PASTE(...) NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_GET_MACRO(__VA_ARGS__, \ @@ -3034,6 +3035,10 @@ void templated_json_throw(ExceptionType exception) // arguments, so dispatching on Type,BaseType,member... directly would run out // one slot early and cap the derived-type macros at 62 members instead of the // 63 that NLOHMANN_JSON_PASTE supports. +// +// The slot table below (down to the closing NLOHMANN_JSON_TYPE_BODY_SENTINEL)) +// is generated by tools/macro_builder (see its README.md; run with the +// "type_body" argument). #define NLOHMANN_JSON_TYPE_BODY(Prefix, ...) NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_GET_MACRO(__VA_ARGS__, \ Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, \ Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, \ diff --git a/tools/amalgamate/README.md b/tools/amalgamate/README.md index 33ea9687b..658e57cbd 100644 --- a/tools/amalgamate/README.md +++ b/tools/amalgamate/README.md @@ -1,8 +1,8 @@ # amalgamate.py - Amalgamate C source and header files -Origin: https://bitbucket.org/erikedlund/amalgamate - -Mirror: https://github.com/edlund/amalgamate +Origin: https://github.com/edlund/amalgamate (formerly hosted at +https://bitbucket.org/erikedlund/amalgamate, which no longer exists; see +`CHANGES.md` for the upstream commit this copy is based on) `amalgamate.py` aims to make it easy to use SQLite-style C source and header amalgamation in projects. @@ -41,21 +41,22 @@ results. ## Installing amalgamate.py -Python v.2.7.0 or higher is required. +Python 3 is required. -`amalgamate.py` can be tested and installed using the following commands: - - ./test.sh && sudo -k cp ./amalgamate.py /usr/local/bin/ +In this repository, `amalgamate.py` is not installed separately; it is run in +place through `make amalgamate`, which calls it once for `json.hpp` and once +for `json_fwd.hpp` (see the root `Makefile`). ## Using amalgamate.py - amalgamate.py [-v] -c path/to/config.json -s path/to/source/dir \ - [-p path/to/prologue.(c|h)] + amalgamate.py -c path/to/config.json -s path/to/source/dir \ + [-p path/to/prologue.(c|h)] [--verbose=yes|no] * The `-c, --config` option should specify the path to a JSON config file which lists the source files, include paths and where to write the resulting - amalgamation. Have a look at `test/source.c.json` and `test/include.h.json` - to see two examples. + amalgamation. `config_json.json` and `config_json_fwd.json` in this + directory are the configs used for `json.hpp` and `json_fwd.hpp`; each + sets `target`, `sources` and `include_paths`. The optional `external` list names include paths that are kept as `#include` directives instead of being inlined, e.g. `["nlohmann/json.hpp"]` for a header @@ -68,3 +69,6 @@ Python v.2.7.0 or higher is required. * The `-p, --prologue` option should specify the path to a file which will be added to the beginning of the amalgamation. It is optional. + * The `-v, --verbose` option takes `yes` or `no` (for example + `--verbose=yes`, as used by the Makefile). It is optional. + diff --git a/tools/generate_natvis/README.md b/tools/generate_natvis/README.md index e11f29eec..3995758c2 100644 --- a/tools/generate_natvis/README.md +++ b/tools/generate_natvis/README.md @@ -2,8 +2,26 @@ Generate the Natvis debugger visualization file for all supported namespace combinations. +The ABI tag list and the library version are parsed from +`include/nlohmann/detail/abi_macros.hpp`, so this script must be re-run (via +`make natvis`) whenever an `NLOHMANN_JSON_ABI_TAG_*` macro is added to that +file or the library version is bumped — otherwise the committed +`nlohmann_json.natvis` drifts from the header it visualizes, and +`make check-amalgamation` fails. + ## Usage ```shell -./generate_natvis.py --version X.Y.Z output_directory/ +make natvis ``` + +or, equivalently: + +```shell +./generate_natvis.py [--version X.Y.Z] [repository_root/] +``` + +`--version` and the output/repository-root directory both default to values +derived from this script's own location, so they only need to be given +explicitly when generating a Natvis file for a different checkout or a +version other than the one in `abi_macros.hpp`. diff --git a/tools/generate_natvis/generate_natvis.py b/tools/generate_natvis/generate_natvis.py index 8562e8f1a..f4cd6fcd2 100755 --- a/tools/generate_natvis/generate_natvis.py +++ b/tools/generate_natvis/generate_natvis.py @@ -7,21 +7,75 @@ import os import re import sys +# Directory of the repository, assuming this script stays at +# tools/generate_natvis/generate_natvis.py. Used only as the default value +# for the "output" argument below. +REPO_ROOT = os.path.normpath(os.path.join(sys.path[0], '..', '..')) + + def semver(v): if not re.fullmatch(r'\d+\.\d+\.\d+', v): raise ValueError return v + +def abi_info(repo_root): + """Derive the ABI tag list (in NLOHMANN_JSON_ABI_TAGS_CONCAT order) and the + library version from /include/nlohmann/detail/abi_macros.hpp, + so this script cannot drift from the header it visualizes.""" + abi_macros_hpp = os.path.join(repo_root, 'include', 'nlohmann', 'detail', 'abi_macros.hpp') + with open(abi_macros_hpp) as f: + content = f.read() + + # find the NLOHMANN_JSON_ABI_TAGS_CONCAT(...) invocation that lists the + # NLOHMANN_JSON_ABI_TAG_* identifiers in order (not its own #define, which + # only names its formal parameters a, b, c, ...) + tag_idents = None + for args in re.findall(r'NLOHMANN_JSON_ABI_TAGS_CONCAT\(\s*(.*?)\)', content, re.S): + idents = re.findall(r'NLOHMANN_JSON_ABI_TAG_\w+', args) + if idents: + tag_idents = idents + break + if not tag_idents: + raise ValueError(f'could not find NLOHMANN_JSON_ABI_TAGS_CONCAT(...) in {abi_macros_hpp}') + + abi_tags = [] + for ident in tag_idents: + # each tag is #define'd to its suffix (e.g. _diag) when the matching + # JSON_* option is enabled, and to nothing in the #else branch; only + # the non-empty definition matches here + match = re.search(r'#define\s+' + re.escape(ident) + r'\s+(_\w+)\s*\n', content) + if not match: + raise ValueError(f'could not find a non-empty #define for {ident} in {abi_macros_hpp}') + abi_tags.append(match.group(1)) + + version = {} + for part in ('MAJOR', 'MINOR', 'PATCH'): + match = re.search(r'#define\s+NLOHMANN_JSON_VERSION_' + part + r'\s+(\d+)', content) + if not match: + raise ValueError(f'could not find NLOHMANN_JSON_VERSION_{part} in {abi_macros_hpp}') + version[part] = match.group(1) + + return abi_tags, '{MAJOR}.{MINOR}.{PATCH}'.format(**version) + + if __name__ == '__main__': parser = argparse.ArgumentParser() - parser.add_argument('--version', required=True, type=semver, help='Library version number') - parser.add_argument('output', help='Output directory for nlohmann_json.natvis') + parser.add_argument('--version', type=semver, + help='Library version number (default: parsed from ' + 'include/nlohmann/detail/abi_macros.hpp below "output")') + parser.add_argument('output', nargs='?', default=REPO_ROOT, + help='Repository root: where include/nlohmann/detail/abi_macros.hpp is ' + 'read from and where nlohmann_json.natvis is written ' + '(default: the repository root this script lives in)') args = parser.parse_args() + derived_tags, derived_version = abi_info(args.output) + namespaces = ['nlohmann'] abi_prefix = 'json_abi' - abi_tags = ['_diag', '_ldvcmp', '_dp', '_bics', '_psp', '_snul'] - version = '_v' + args.version.replace('.', '_') + abi_tags = derived_tags + version = '_v' + (args.version or derived_version).replace('.', '_') inline_namespaces = [] # generate all combinations of inline namespace names diff --git a/tools/macro_builder/README.md b/tools/macro_builder/README.md new file mode 100644 index 000000000..f425f22c9 --- /dev/null +++ b/tools/macro_builder/README.md @@ -0,0 +1,51 @@ +# macro_builder + +Generates the argument-counting macros behind the `NLOHMANN_DEFINE_TYPE_*` and `NLOHMANN_DEFINE_DERIVED_TYPE_*` +macros in [`include/nlohmann/detail/macro_scope.hpp`](../../include/nlohmann/detail/macro_scope.hpp): + +- `NLOHMANN_JSON_EXPAND` +- `NLOHMANN_JSON_GET_MACRO`, which selects a macro by the number of its arguments (64 slots) +- `NLOHMANN_JSON_PASTE`, which calls a function-like macro for each member, and its helpers `NLOHMANN_JSON_PASTE2` + to `NLOHMANN_JSON_PASTE64` +- `NLOHMANN_JSON_DOUBLE_PASTE`, which the `*_WITH_NAMES` macros use to call a function-like macro for each + (JSON name, member) pair, and its helpers `NLOHMANN_JSON_DOUBLE_PASTE3` to `NLOHMANN_JSON_DOUBLE_PASTE63` +- the slot table of `NLOHMANN_JSON_TYPE_BODY`, which dispatches `NLOHMANN_DEFINE_TYPE_*(Type)` (no further + arguments) to the zero-member implementation and every other argument count to the one-or-more-member + implementation + +The number of slots (`max_args` in [`main.cpp`](main.cpp)) sets the member limit of these macros. +`NLOHMANN_JSON_PASTE` and `NLOHMANN_JSON_TYPE_BODY` take the function/prefix as their first argument, so 64 slots +allow 63 members; `NLOHMANN_JSON_DOUBLE_PASTE` additionally consumes its members two at a time (name, member), so +it only defines the odd helpers up to `NLOHMANN_JSON_DOUBLE_PASTE63`. + +## Usage + +From the project root: + +```shell +c++ -std=c++11 tools/macro_builder/main.cpp -o macro_builder +./macro_builder +./macro_builder type_body +``` + +1. Run `./macro_builder` (no arguments). In `include/nlohmann/detail/macro_scope.hpp`, replace the lines from + `#define NLOHMANN_JSON_EXPAND( x ) x` to the `#define NLOHMANN_JSON_DOUBLE_PASTE63(...)` line with the output, + without its trailing empty line. +2. Run `./macro_builder type_body`. Replace the lines from `#define NLOHMANN_JSON_TYPE_BODY(Prefix, ...)` to the + `NLOHMANN_JSON_TYPE_BODY_SENTINEL))` line with the output. +3. Run `make amalgamate`. It updates `single_include/nlohmann/json.hpp` and runs `make pretty`, which indents the + continuation lines that the tool writes unindented. + +With an unchanged `main.cpp`, these steps reproduce both blocks of `macro_scope.hpp` byte for byte. `make +macro_builder_check` (also run by CI, see `.github/workflows/check_amalgamation.yml`) automates this: it builds +`main.cpp`, regenerates both blocks, and fails on a diff against the checked-in header. + +## Maintained by hand + +The tool does not generate everything that depends on the number of slots. When changing `max_args`, also update: + +- the documented limit of 63 members in `docs/mkdocs/docs` and the tests at that limit in + `tests/src/unit-udt_macro.cpp` + +All three tables pass one macro name per slot to `NLOHMANN_JSON_GET_MACRO`, so they need exactly as many entries as +it has slots. diff --git a/tools/macro_builder/main.cpp b/tools/macro_builder/main.cpp index e676daacc..f035048f2 100644 --- a/tools/macro_builder/main.cpp +++ b/tools/macro_builder/main.cpp @@ -1,10 +1,14 @@ #include #include #include +#include using namespace std; -void build_code(int max_args) +// Builds NLOHMANN_JSON_EXPAND, NLOHMANN_JSON_GET_MACRO, and the +// NLOHMANN_JSON_PASTE / NLOHMANN_JSON_PASTE2..PASTE dispatch table +// and recursive definitions. +string build_paste_code(int max_args) { stringstream ss; ss << "#define NLOHMANN_JSON_EXPAND( x ) x" << endl; @@ -12,32 +16,93 @@ void build_code(int max_args) for (int i = 0 ; i < max_args ; i++) ss << "_" << i + 1 << ", "; ss << "NAME,...) NAME" << endl; - + ss << "#define NLOHMANN_JSON_PASTE(...) NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_GET_MACRO(__VA_ARGS__, \\" << endl; for (int i = max_args ; i > 1 ; i--) ss << "NLOHMANN_JSON_PASTE" << i << ", \\" << endl; ss << "NLOHMANN_JSON_PASTE1)(__VA_ARGS__))" << endl; - + ss << "#define NLOHMANN_JSON_PASTE2(func, v1) func(v1)" << endl; for (int i = 3 ; i <= max_args ; i++) { - ss << "#define NLOHMANN_JSON_PASTE" << i << "(func, "; + ss << "#define NLOHMANN_JSON_PASTE" << i << "(func, "; for (int j = 1 ; j < i -1 ; j++) - ss << "v" << j << ", "; + ss << "v" << j << ", "; ss << "v" << i-1 << ") NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE" << i-1 << "(func, "; for (int j = 2 ; j < i-1 ; j++) ss << "v" << j << ", "; ss << "v" << i-1 << ")" << endl; } - - cout << ss.str() << endl; + + return ss.str(); } -int main(int argc, char** argv) +// Builds the NLOHMANN_JSON_DOUBLE_PASTE dispatch table and recursive +// definitions used by the *_WITH_NAMES macros. Its GET_MACRO dispatch reuses +// the same max_args slots as NLOHMANN_JSON_PASTE, but DOUBLE_PASTE consumes +// its arguments two at a time (name, member), so an even slot count falls +// back to the next lower odd NLOHMANN_JSON_DOUBLE_PASTE. +string build_double_paste_code(int max_args) +{ + stringstream ss; + ss << "#define NLOHMANN_JSON_DOUBLE_PASTE(...) NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_GET_MACRO(__VA_ARGS__, \\" << endl; + for (int i = max_args ; i > 1 ; i--) + { + int k = (i % 2 == 1) ? i : i - 1; + ss << "NLOHMANN_JSON_DOUBLE_PASTE" << k << ", \\" << endl; + } + ss << "NLOHMANN_JSON_DOUBLE_PASTE1)(__VA_ARGS__))" << endl; + + ss << "#define NLOHMANN_JSON_DOUBLE_PASTE3(func, v1, v2) func(v1, v2)" << endl; + for (int k = 5 ; k <= max_args - 1 ; k += 2) + { + ss << "#define NLOHMANN_JSON_DOUBLE_PASTE" << k << "(func, "; + for (int j = 1 ; j < k - 1 ; j++) + ss << "v" << j << ", "; + ss << "v" << k - 1 << ") NLOHMANN_JSON_DOUBLE_PASTE3(func, v1, v2) NLOHMANN_JSON_DOUBLE_PASTE" << k - 2 << "(func, "; + for (int j = 3 ; j < k - 1 ; j++) + ss << "v" << j << ", "; + ss << "v" << k - 1 << ")" << endl; + } + + return ss.str(); +} + +// Builds the NLOHMANN_JSON_TYPE_BODY dispatch table: max_args - 1 slots +// selecting the *_MEMBERS implementation and a final slot selecting +// *_EMPTY, so NLOHMANN_DEFINE_TYPE_*(Type) with no further arguments still +// resolves (issue #4041). +string build_type_body_table(int max_args) +{ + stringstream ss; + ss << "#define NLOHMANN_JSON_TYPE_BODY(Prefix, ...) NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_GET_MACRO(__VA_ARGS__, \\" << endl; + const int per_line = 8; + for (int i = 1 ; i <= max_args ; i++) + { + ss << (i == max_args ? "Prefix ## EMPTY" : "Prefix ## MEMBERS") << ", "; + if (i % per_line == 0) + ss << "\\" << endl; + } + ss << "NLOHMANN_JSON_TYPE_BODY_SENTINEL))" << endl; + + return ss.str(); +} + +int main(int argc, char** argv) { int max_args = 64; - build_code(max_args); - + + // With "type_body", print only the NLOHMANN_JSON_TYPE_BODY dispatch + // table (a separate insertion point in macro_scope.hpp); otherwise + // print the EXPAND/GET_MACRO/PASTE/DOUBLE_PASTE block that precedes it. + if (argc > 1 && string(argv[1]) == "type_body") + { + cout << build_type_body_table(max_args); + } + else + { + cout << build_paste_code(max_args) << build_double_paste_code(max_args); + } + return 0; } - diff --git a/tools/serve_header/serve_header.py b/tools/serve_header/serve_header.py index 1f29cb589..5b5bf03f4 100755 --- a/tools/serve_header/serve_header.py +++ b/tools/serve_header/serve_header.py @@ -5,6 +5,8 @@ import logging import os import re import shutil +import socket +import ssl import sys import subprocess @@ -34,7 +36,7 @@ JSON_VERSION_RE = re.compile(r'\s*#\s*define\s+NLOHMANN_JSON_VERSION_MAJOR\s+') class ExitHandler(logging.StreamHandler): def __init__(self, level): - """.""" + """Exit the process on log records at or above level.""" super().__init__() self.level = level @@ -54,7 +56,7 @@ def is_project_root(test_dir='.'): class DirectoryEventBucket: def __init__(self, callback, delay=1.2, threshold=0.8): - """.""" + """Batch directory events and pass their common path to callback.""" self.delay = delay self.threshold = timedelta(seconds=threshold) self.callback = callback @@ -99,7 +101,7 @@ class WorkTree: make_command = 'make' def __init__(self, root_dir, tree_dir): - """.""" + """Track the working tree at tree_dir and its amalgamated header.""" self.root_dir = root_dir self.tree_dir = tree_dir self.rel_dir = os.path.relpath(tree_dir, root_dir) @@ -114,11 +116,11 @@ class WorkTree: self.build_time = t.strftime(DATETIME_FORMAT) def __hash__(self): - """.""" + """Hash by working tree directory.""" return hash((self.tree_dir)) def __eq__(self, other): - """.""" + """Compare by working tree directory.""" if not isinstance(other, type(self)): return NotImplemented return self.tree_dir == other.tree_dir @@ -150,7 +152,7 @@ class WorkTree: class WorkTrees(FileSystemEventHandler): def __init__(self, root_dir): - """.""" + """Find the working trees below root_dir and watch it for changes.""" super().__init__() self.root_dir = root_dir self.trees = set([]) @@ -250,11 +252,11 @@ class WorkTrees(FileSystemEventHandler): self.observer.stop() self.observer.join() -class HeaderRequestHandler(SimpleHTTPRequestHandler): # lgtm[py/missing-call-to-init] +class HeaderRequestHandler(SimpleHTTPRequestHandler): cors_origins = DEFAULT_CORS_ORIGINS def __init__(self, request, client_address, server): - """.""" + """Handle a request for a header below the working trees' root directory.""" self.worktrees = server.worktrees self.worktree = None try: @@ -336,7 +338,7 @@ class HeaderRequestHandler(SimpleHTTPRequestHandler): # lgtm[py/missing-call-to- class DualStackServer(ThreadingHTTPServer): def __init__(self, addr, worktrees): - """.""" + """Serve the headers of worktrees on addr.""" self.worktrees = worktrees super().__init__(addr, HeaderRequestHandler) @@ -349,8 +351,6 @@ class DualStackServer(ThreadingHTTPServer): if __name__ == '__main__': import argparse - import ssl - import socket import yaml # exit code From c261578431247df036b4d7c221943cde16e21d1f Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Wed, 30 Sep 2026 22:48:32 +0200 Subject: [PATCH 7/7] Deduplicate binary reader/writer helpers and fix stale comments (#5730) * Fix stale and missing comments in binary_writer The doc block of write_number() ended up above the byte_swap() helpers added in #5286, about 80 lines from the function. It was also a plain comment that Doxygen skips, said "write a number to output input", and left BON8 out of the big-endian formats. Move it back onto write_number() as a /*! block and fix the text. write_bson() documented "@pre j.type() == value_t::object", but it throws type_error.317 for every other type, and to_bson() relies on that. Document the exception instead. Explain why the CBOR binary subtype is always written with a 0xD8..0xDB head and never in the one-byte tag form: binary_reader with cbor_tag_handler_t::store only keeps those heads as a subtype, so switching to write_cbor_head() would break round trips for subtypes 0..23. Also fix the grammar of the to_char_type comment. Comments only; no change in behavior, API or ABI. Part of #5710 Signed-off-by: Niels Lohmann * Merge the duplicated UBJSON/BJData integer marker ladders write_number_with_ubjson_prefix() (unsigned and signed overloads) and ubjson_prefix() (number_integer and number_unsigned cases) each picked the UBJSON/BJData integer marker (i, U, I, u, l, m, L, M, H) with their own independent if/else ladder, and the values beyond 64 bits were handled by a second, tag-dispatched pair of ladders. An optimized container announces the marker of its first element via ubjson_prefix() and then writes every element through write_number_with_ubjson_prefix(), so the two had to be kept in lockstep by hand across four call sites. Replace all of that with one ubjson_integer_prefix() built on value_in_range_of, and one write_ubjson_integer_payload() that writes the value (or, for 'H', the decimal digits) for a given marker. write_number_with_ubjson_prefix() and ubjson_prefix() keep their signatures and now just call these two helpers. Behavior, the public API and the ABI are unchanged. Verified with a new regression test covering scalars and $-optimized arrays/objects at every int8/uint8/int16/uint16/int32/uint32/int64/uint64 boundary for to_ubjson/to_bjdata (both use_size/use_type settings), and by diffing to_ubjson/to_bjdata output before and after over the json_test_data corpus (bit-identical). Part of #5710 Signed-off-by: Niels Lohmann * Remove dead get_char parameters in binary_reader The non-recursive rewrite of the binary readers (#5505, #5506, #5507) left parse_cbor_internal()'s and parse_ubjson_internal()'s get_char parameters dead: parse_cbor_internal() has one caller and it always passes true, and parse_ubjson_internal() has one caller and it always uses the true default. Both parameters, and the @param docs describing the "reuse the last character" mode they used to select, no longer correspond to anything. Drop both parameters, initialise fetch/prefix unconditionally, and update the two call sites in sax_parse(). parse_cbor_value()'s and get_ubjson_string()'s own get_char parameters are unrelated and are left alone; both still have a false caller. Also delete a stray `@return whether a valid MessagePack value was passed to the SAX parser` doxygen block that sits directly above parse_msgpack_value()'s real doc comment, a leftover of the same rewrite. Behavior, the public API and the ABI are unchanged; these are private members of detail::binary_reader. Verified by compiling with -Wunused-parameter and running unit-cbor, unit-ubjson, unit-bjdata and unit-msgpack (offline, against the stubbed test_data.hpp). Part of #5711 Signed-off-by: Niels Lohmann * Share the IEEE half-precision decoder between CBOR and BJData binary_reader had two ~45-line copies of the IEEE 754 half-precision decoder: CBOR's case 0xF9 and BJData's case 'h'. Once formatting is normalised, the two blocks were identical except for the byte order used to assemble the 16-bit half (CBOR is big endian, BJData is little endian). Any future change to half-float decoding had to be made and kept in sync in both places. Add one get_half_float(format, little_endian) helper that does the two get()/unexpect_eof() reads, assembles the half in the requested byte order, decodes it per RFC 8949 Appendix D, and calls sax->number_float. Both cases now just call it with their byte order; the BJData case keeps its bjdata-only guard. Behavior, the public API and the ABI are unchanged. Verified with a scratch probe comparing the old and new decoders bit-for-bit (NaN by isnan()) over all 65536 wire byte pairs, in both formats, and by running unit-cbor and unit-bjdata (offline, against the stubbed test_data.hpp). Part of #5711 Signed-off-by: Niels Lohmann * Deduplicate the MessagePack unsigned-integer writer ladder The number_integer (non-negative branch) and number_unsigned cases in write_msgpack() each held their own copy of the fixint/uint8/16/32/64 ladder, kept in lockstep only by a comment ("we used the code from the value_t::number_unsigned case here"). Both copies mixed union members: the signed copy compared number_unsigned but wrote number_integer, and vice versa. Extract write_msgpack_unsigned(std::uint64_t), mirroring how write_cbor_head() already avoids the same duplication for CBOR, and call it from both cases. Each case now reads only its own active union member. Output bytes are unchanged for the default 64-bit number types. #5710 item 3 Signed-off-by: Niels Lohmann * Unify float marker selection and fix the long double compile error Four formats picked between a float32 and float64 marker through four different helper styles: dummy-argument overloads for CBOR and MessagePack, an std::is_same template for BON8, and a runtime if-chain on input_format_t for write_compact_float(). With number_float_t set to long double, to_cbor, to_msgpack and to_ubjson failed inside the library with "call to 'get_cbor_float_prefix' is ambiguous", while to_bson kept working because write_bson_double() takes a plain double. Change write_compact_float() to take the two marker bytes directly (each of its three callers already knows them at compile time) instead of an input_format_t it only forwarded, and delete the now-unused get_cbor_float_prefix(), get_msgpack_float_prefix(), get_bon8_float_prefix() and get_compact_float_prefix() helpers. Turn the two get_ubjson_float_prefix() overloads into one template. Both write_compact_float() and get_ubjson_float_prefix() now report an unsupported number_float_t with a static_assert naming the requirement, rather than an ambiguous-overload error; the assert lives in the function body, not the class scope, so to_bson with long double is unaffected. Verified with a probe basic_json<..., long double>: to_bson still compiles and round-trips, while to_cbor/to_msgpack/to_ubjson now fail to compile with the new static_assert message. This changes the text of an existing compile error for users with an unsupported number_float_t (documented as a public-API-visible change in #5710). #5710 item 1 Signed-off-by: Niels Lohmann * Deduplicate the BJData ndarray writer's dtype dispatch and drop write_bjdata_ndarray() built a 12-entry std::map on every call just to translate the _ArrayType_ name to a dtype marker (the only reason binary_writer.hpp included ), then mapped dtype to C++ type twice more: once as a switch for the range-check pass and once as a separate if/else chain for the write pass, with nothing checking that the two agreed. The caller also ran three at() lookups, and the callee called value.at(key) about ten more times for the same three members. Replace the map with bjdata_ndarray_type_marker(), a plain string comparison chain (a C++11 constexpr function cannot contain a switch, so this mirrors binary_reader's own static table style). Replace the switch/if-chain pair with one write_bjdata_ndarray_elements() that switches on dtype once and calls a per-type helper - write_bjdata_ndarray_element() for the eight integer dtypes and write_bjdata_ndarray_float_element() for 'd' - with a dry_run flag selecting the range check or the actual write, so the two passes can no longer disagree on the type. _ArrayType_, _ArraySize_ and _ArrayData_ are now looked up once into references, and the four header marker bytes ('[', '$', '#') are written through to_char_type() like the rest of the UBJSON/BJData writer. The 'd' (single-precision) rule is left exactly as before, since #5707 is expected to change it separately. Verified byte-for-byte identical output before/after for every dtype (including the Draft 2/Draft 3 'byte' fallback and the use_count/ use_type combinations) via a standalone probe, plus round-tripping through from_bjdata(). Overlaps #5707, which is expected to touch the 'd' dtype case, and #5518, which is expected to move the write_bjdata_ndarray() call site. #5710 item 4 Signed-off-by: Niels Lohmann * Assert that write_bson_document() consumes every calc_bson_sizes() entry calc_bson_sizes() and write_bson_document() are a hand-synchronized pair of passes over the same object/array tree, introduced by #5553: the size pass appends to nested_sizes in visiting order, and the write pass consumes the table by position with nested_sizes[next_size++]. Nothing checked that the write pass consumed the whole table. If a future change touched only one of the two passes - for example to skip or reject an entry - every later size prefix in the document would be silently wrong. Add JSON_ASSERT(next_size == nested_sizes.size()) where write_bson_document() returns, so such a future drift between the two passes is caught immediately (JSON_ASSERT expands to nothing in release builds using assert(), and the fuzzers/tests already build with it enabled). The two passes agree today, so this changes nothing observable; it only guards against the risk described in #5710 item 5. Extracting a shared stepper for the two passes (the second half of the proposed change) is left for a follow-up: it only saves ~30 lines and the issue asks for it only if the result reads clearly, which needs more room to get right than a mechanical cleanup pass allows. #5710 item 5 Signed-off-by: Niels Lohmann * Make the BJData lookup tables static functions instead of members binary_reader held bjd_optimized_type_markers and bjd_types_map as non-static const members (12 string_t objects for the type-name table), built and destroyed on every from_cbor/from_msgpack/from_bson/ from_ubjson/from_bon8/from_bjdata call even though only from_bjdata ever reads them. They also needed the #define/decltype/#undef workaround from #3637 and two NOLINTNEXTLINE suppressions, and binary_writer already carries the same two lists in another form (is_bjdata_excluded_type_marker() and a local std::map in write_bjdata_ndarray(), the latter removed by the item-4 commit), so the excluded-marker lists could drift apart. Replace bjd_optimized_type_markers with static constexpr is_bjd_excluded_optimized_type(char_int_type), using the same ||-chain as binary_writer's is_bjdata_excluded_type_marker(). Replace bjd_types_map with a non-constexpr static bjd_type_name(char_int_type) switch returning nullptr for an unknown marker (a C++11 constexpr function cannot contain a switch). Delete both JSON_BINARY_READER_MAKE_* macros, the bjd_type pair alias, the NOLINTNEXTLINE suppressions, detail::make_array() (no longer used anywhere), and the now-unused and includes. Update the two call sites (the ND-array excluded-type check and the _ArrayType_ lookup) accordingly, and replace unit-bjdata.cpp's "LUT arrays are sorted" section, which only checked the two tables' internal ordering, with a check of all 12 type names and all 8 excluded markers against both new functions. #5711 item 1 Signed-off-by: Niels Lohmann * Read CBOR's 1/2/4/8-byte argument through one helper parse_cbor_internal() hand-wrote the same "read a 1/2/4/8-byte big-endian unsigned integer" ladder four times over: - twice for tag numbers 0xD8-0xDB, once in the tag_handler::ignore branch and once, nearly identically, in the ::store branch (~90 lines to read one integer); - twice more for container lengths, once for array heads 0x98-0x9B and once for map heads 0xB8-0xBB, where the 1/2-byte forms called enter_array()/enter_object() directly and the 4/8-byte forms additionally went through get_cbor_container_size(). Add get_cbor_argument(std::uint64_t&), reading the width selected by current & 0x1F via the same get_number() calls as before (so EOF is reported exactly as before), and route all four sites through it: - 0xD8-0xDB now read the argument once per branch instead of switching on `current` a second time; behavior split cleanly from embedded tags 0xC0-0xD7 (tag value in the head, no argument to read), which is now its own case block that no longer has to fall into the ::store switch's "default" case to reach the same tag_pending = true; return true; outcome. - 0x98-0x9B and 0xB8-0xBB collapse into one case block each, always going through get_cbor_container_size() (harmless for 1/2-byte lengths, which already always fit). Verified byte-for-byte identical behavior before/after with a standalone probe covering embedded and multi-byte tags under all three tag_handler_t settings, a tag over a byte string (subtype path), truncated tag/length arguments of every width, and array/map lengths of every width, including the out_of_range.408 "excessive size" case: same exceptions, same messages, same chars_read, same successful results. Left the string/byte-string length ladders in get_cbor_string()/ get_cbor_binary() untouched, as noted in #5711 item 2, since #5325 is expected to touch them separately. Overlaps #5601 (adds a branch right above the embedded-tag case) and #5607 (touches the integer cases 0x18-0x1B, which share this ladder's shape in separate hunks). #5711 item 2 Signed-off-by: Niels Lohmann * Add leave_container() to match enter_container() Every container is opened through enter_container(), whose docs promise that a check placed there runs before every start event. The close side had no equivalent: the same "container_stack.pop_back(); dispatch to end_object() or end_array()" sequence was written out separately in BSON, CBOR, MessagePack, UBJSON/BJData and BON8, each copying the pattern of keeping an is_object flag around the pop_back() that would otherwise invalidate a reference to it. A check needed on close would have had to be added in five places, and a sixth copy could go unnoticed. Add leave_container() next to enter_container(), doing the same pop-then-dispatch, and replace the five sites with it. Each site keeps its own surrounding logic (BSON's check_bson_document_size() call before popping, MessagePack's is_object copy used again below, UBJSON/BJData's remaining-container handling after popping, BON8's top used again below); only the repeated pop/dispatch line pair is now shared. Verified all six binary-format unit suites and unit-regression2's deep-nesting tests (dependent count/reuse count and the bjdata ndarray depth cases) still pass, compiled with -Wall -Wextra and ASan/UBSan. Overlaps #5601, which is expected to add a sixth close site in its own skip loop; that site can route through leave_container() too once it lands. #5711 item 4 Signed-off-by: Niels Lohmann * Stop passing the input format to sax_parse() when the reader already has it binary_reader's constructor stores the format in the input_format member, and sax_parse(format, sax_, strict, tag_handler) took the same value again purely to dispatch on it. Every in-tree caller passed the same value both times (all 16 from_cbor/from_msgpack/from_ubjson/ from_bjdata/from_bon8/from_bson call sites in json.hpp, and the three public basic_json::sax_parse() overloads), so nothing was broken today, but a caller of the detail class directly (only reachable via JSON_PRIVATE_UNLESS_TESTED, as unit-bjdata.cpp already does) could pass a mismatched pair - say bjdata to the constructor and ubjson to sax_parse - and dispatch on one format while applying the other format's rules; the default-constructed input_format_t::json reader would additionally hit JSON_ASSERT(false) in exception_message() on its first error. Add sax_parse(json_sax_t*, bool, cbor_tag_handler_t) forwarding to the existing overload with the stored input_format, and switch every caller to it: the 16 from_*() sites (keeping their `// cppcheck-suppress[accessMoved]` comments) and the three basic_json::sax_parse() overloads, all of which already had the format available from their own `format` parameter. The four-argument overload is kept for anyone still calling it, now with JSON_ASSERT(format == input_format) so a mismatch fails immediately in a debug build (assert-enabled binaries, including the fuzzers and test suite) instead of misbehaving; verified with a probe that constructs a reader for one format and calls the explicit overload with another, which aborts on that assertion as expected. Removing or asserting against the constructor's input_format_t::json default, which would affect direct detail users, is left as a separate decision per #5711 item 5. Overlaps #5601, which is expected to add an AllowRecovery template parameter to sax_parse() and touch these same call sites in json.hpp. #5711 item 5 Signed-off-by: Niels Lohmann * Deduplicate UBJSON/BJData signed-count handling, drop dead ndarray checks get_ubjson_size_value()'s 'i'/'I'/'l'/'L' cases each read a differently sized signed integer and then repeated the same "reject negative with error 113" check; only 'L' additionally checked value_in_range_of for the out_of_range.408 case. Any change to that error path had to be made four times. Add get_ubjson_signed_count(std::size_t&), doing the read, the negative check and the range check once, and route all four markers through it. The range check is a no-op for 'i'/'I'/'l' (their values always fit std::size_t) and only live for 'L' on a 32-bit std::size_t target, matching today's behavior exactly. In the ndarray dimension-product loop, the preceding loop already returns early on any zero dimension and result starts at 1, so `i > 0` in the pre-multiplication overflow check was always true, and `result == 0` in the post-multiplication check could not be reached either: two positive factors whose product does not overflow (as the pre-check already guarantees) cannot be zero. Drop the dead `i > 0 &&` and narrow the post-check to `result == npos`, the one case the pre-check cannot rule out (an exact, non-overflowing match with the sentinel reserved for unknown-size containers), with a comment explaining why. Verified byte-for-byte identical behavior before/after with a standalone probe covering negative counts for every marker, a matching positive count, and ndarray inputs, plus the full unit-ubjson and unit-bjdata suites (same assertion counts as before this change). Overlaps #5601 (rewrites the four parse_error calls and the overflow checks touched here) and #5607/#5707 (touch neighboring lines in the same functions). #5711 item 6 Signed-off-by: Niels Lohmann * Drop redundant format parameter and dummy float argument (review) binary_reader::sax_parse(format, ...) only ever had to equal the format given to the constructor, which it asserted. With every caller already on the format-less overload, remove the four-argument overload and dispatch on the stored input_format directly. binary_reader is a detail class, so this is not a public API change. get_ubjson_float_prefix() took a value only to deduce its type; make the type an explicit template argument instead. Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- .../nlohmann/detail/input/binary_reader.hpp | 652 ++++--- include/nlohmann/detail/meta/cpp_future.hpp | 7 - .../nlohmann/detail/output/binary_writer.hpp | 882 ++++----- include/nlohmann/json.hpp | 38 +- single_include/nlohmann/json.hpp | 1579 +++++++---------- tests/src/unit-bjdata.cpp | 33 +- tests/src/unit-ubjson.cpp | 221 +++ 7 files changed, 1623 insertions(+), 1789 deletions(-) diff --git a/include/nlohmann/detail/input/binary_reader.hpp b/include/nlohmann/detail/input/binary_reader.hpp index b9e6b304b..a4dcdb584 100644 --- a/include/nlohmann/detail/input/binary_reader.hpp +++ b/include/nlohmann/detail/input/binary_reader.hpp @@ -8,7 +8,6 @@ #pragma once -#include // generate_n #include // array #include // ldexp #include // size_t @@ -123,16 +122,16 @@ class binary_reader ~binary_reader() = default; /*! - @param[in] format the binary format to parse + @brief parse in the format the constructor was given + @param[in] sax_ a SAX event processor @param[in] strict whether to expect the input to be consumed completed @param[in] tag_handler how to treat CBOR tags @return whether parsing was successful */ - JSON_HEDLEY_NON_NULL(3) - bool sax_parse(const input_format_t format, - json_sax_t* sax_, + JSON_HEDLEY_NON_NULL(2) + bool sax_parse(json_sax_t* sax_, const bool strict = true, const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error) { @@ -141,14 +140,14 @@ class binary_reader bon8_pushback_size = 0; bool result = false; - switch (format) + switch (input_format) { case input_format_t::bson: result = parse_bson_internal(); break; case input_format_t::cbor: - result = parse_cbor_internal(true, tag_handler); + result = parse_cbor_internal(tag_handler); break; case input_format_t::msgpack: @@ -271,6 +270,22 @@ class binary_reader return enter_container(/*is_object*/true, len, type_marker); } + /*! + @brief close the innermost open array or object + + Pops the container opened by the matching @ref enter_container call and + emits the SAX end event. Every format-specific driver otherwise repeated + the same pop-then-dispatch sequence at its own close site. + + @return whether the SAX parser accepted the end event + */ + bool leave_container() + { + const bool is_object = container_stack.back().is_object; + container_stack.pop_back(); + return is_object ? sax->end_object() : sax->end_array(); + } + ////////// // BSON // ////////// @@ -355,8 +370,8 @@ class binary_reader if (element_type == 0) // end of the innermost document { // a copy, not a reference: it must stay valid across the - // pop_back() below, which destroys the container_stack - // element it would otherwise alias + // pop_back() inside leave_container() below, which destroys + // the container_stack element it would otherwise alias const container_frame top = container_stack.back(); if (JSON_HEDLEY_UNLIKELY(!check_bson_document_size(top.start_position, top.declared_size))) @@ -364,8 +379,7 @@ class binary_reader return false; } - container_stack.pop_back(); - if (JSON_HEDLEY_UNLIKELY(top.is_object ? !sax->end_object() : !sax->end_array())) + if (JSON_HEDLEY_UNLIKELY(!leave_container())) { return false; } @@ -860,29 +874,13 @@ class binary_reader return enter_array(conditional_static_cast(static_cast(current) & 0x1Fu)); case 0x98: // array (one-byte uint8_t for n follows) - { - std::uint8_t len{}; - return get_number(input_format_t::cbor, len) && enter_array(static_cast(len)); - } - case 0x99: // array (two-byte uint16_t for n follow) - { - std::uint16_t len{}; - return get_number(input_format_t::cbor, len) && enter_array(static_cast(len)); - } - case 0x9A: // array (four-byte uint32_t for n follow) - { - std::uint32_t len{}; - std::size_t size{}; - return get_number(input_format_t::cbor, len) && get_cbor_container_size(len, size, "array") && enter_array(size); - } - case 0x9B: // array (eight-byte uint64_t for n follow) { std::uint64_t len{}; std::size_t size{}; - return get_number(input_format_t::cbor, len) && get_cbor_container_size(len, size, "array") && enter_array(size); + return get_cbor_argument(len) && get_cbor_container_size(len, size, "array") && enter_array(size); } case 0x9F: // array (indefinite length) @@ -916,35 +914,19 @@ class binary_reader return enter_object(conditional_static_cast(static_cast(current) & 0x1Fu)); case 0xB8: // map (one-byte uint8_t for n follows) - { - std::uint8_t len{}; - return get_number(input_format_t::cbor, len) && enter_object(static_cast(len)); - } - case 0xB9: // map (two-byte uint16_t for n follow) - { - std::uint16_t len{}; - return get_number(input_format_t::cbor, len) && enter_object(static_cast(len)); - } - case 0xBA: // map (four-byte uint32_t for n follow) - { - std::uint32_t len{}; - std::size_t size{}; - return get_number(input_format_t::cbor, len) && get_cbor_container_size(len, size, "map") && enter_object(size); - } - case 0xBB: // map (eight-byte uint64_t for n follow) { std::uint64_t len{}; std::size_t size{}; - return get_number(input_format_t::cbor, len) && get_cbor_container_size(len, size, "map") && enter_object(size); + return get_cbor_argument(len) && get_cbor_container_size(len, size, "map") && enter_object(size); } case 0xBF: // map (indefinite length) return enter_object(detail::unknown_size()); - case 0xC0: // tagged item + case 0xC0: // tagged item (tag value 0-23, in the head itself) case 0xC1: case 0xC2: case 0xC3: @@ -968,6 +950,22 @@ class binary_reader case 0xD5: case 0xD6: case 0xD7: + { + if (tag_handler == cbor_tag_handler_t::error) + { + auto last_token = get_token_string(); + return sax->parse_error(chars_read, last_token, parse_error::create(112, chars_read, + exception_message(input_format_t::cbor, concat("invalid byte: 0x", last_token), "value"), nullptr)); + } + + // ignore and store: the tag value is already in the head, so + // there is nothing left to read here; the tagged value that + // follows is read by the loop in parse_cbor_internal() rather + // than by recursing here + tag_pending = true; + return true; + } + case 0xD8: // tagged item (1 byte follows) case 0xD9: // tagged item (2 bytes follow) case 0xDA: // tagged item (4 bytes follow) @@ -984,47 +982,11 @@ class binary_reader case cbor_tag_handler_t::ignore: { - // ignore binary subtype - switch (current) + // ignore the tag's binary subtype argument + std::uint64_t subtype_to_ignore{}; + if (!get_cbor_argument(subtype_to_ignore)) { - case 0xD8: - { - std::uint8_t subtype_to_ignore{}; - if (!get_number(input_format_t::cbor, subtype_to_ignore)) - { - return false; - } - break; - } - case 0xD9: - { - std::uint16_t subtype_to_ignore{}; - if (!get_number(input_format_t::cbor, subtype_to_ignore)) - { - return false; - } - break; - } - case 0xDA: - { - std::uint32_t subtype_to_ignore{}; - if (!get_number(input_format_t::cbor, subtype_to_ignore)) - { - return false; - } - break; - } - case 0xDB: - { - std::uint64_t subtype_to_ignore{}; - if (!get_number(input_format_t::cbor, subtype_to_ignore)) - { - return false; - } - break; - } - default: - break; + return false; } // the tagged value follows; it is read by the loop in // parse_cbor_internal() rather than by recursing here @@ -1034,57 +996,15 @@ class binary_reader case cbor_tag_handler_t::store: { - binary_t b; // use binary subtype and store in a binary container - switch (current) + std::uint64_t subtype{}; + if (!get_cbor_argument(subtype)) { - case 0xD8: - { - std::uint8_t subtype{}; - if (!get_number(input_format_t::cbor, subtype)) - { - return false; - } - b.set_subtype(detail::conditional_static_cast(subtype)); - break; - } - case 0xD9: - { - std::uint16_t subtype{}; - if (!get_number(input_format_t::cbor, subtype)) - { - return false; - } - b.set_subtype(detail::conditional_static_cast(subtype)); - break; - } - case 0xDA: - { - std::uint32_t subtype{}; - if (!get_number(input_format_t::cbor, subtype)) - { - return false; - } - b.set_subtype(detail::conditional_static_cast(subtype)); - break; - } - case 0xDB: - { - std::uint64_t subtype{}; - if (!get_number(input_format_t::cbor, subtype)) - { - return false; - } - b.set_subtype(detail::conditional_static_cast(subtype)); - break; - } - default: - { - // as above, the tagged value is read by the caller - tag_pending = true; - return true; - } + return false; } + binary_t b; + b.set_subtype(detail::conditional_static_cast(subtype)); + get(); // a byte string (the heads accepted by get_cbor_binary) keeps the tag as subtype if ((current >= 0x40 && current <= 0x5B) || current == 0x5F) @@ -1115,52 +1035,7 @@ class binary_reader return sax->null(); case 0xF9: // Half-Precision Float (two-byte IEEE 754) - { - const auto byte1_raw = get(); - if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format_t::cbor, "number"))) - { - return false; - } - const auto byte2_raw = get(); - if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format_t::cbor, "number"))) - { - return false; - } - - const auto byte1 = static_cast(byte1_raw); - const auto byte2 = static_cast(byte2_raw); - - // Code from RFC 8949, Appendix D, Figure 3: - // As half-precision floating-point numbers were only added - // to IEEE 754 in 2008, today's programming platforms often - // still only have limited support for them. It is very - // easy to include at least decoding support for them even - // without such support. An example of a small decoder for - // half-precision floating-point numbers in the C language - // is shown in Fig. 3. - const auto half = static_cast((byte1 << 8u) + byte2); - const double val = [&half] - { - const int exp = (half >> 10u) & 0x1Fu; - const unsigned int mant = half & 0x3FFu; - JSON_ASSERT(exp <= 31); - JSON_ASSERT(mant <= 1023); - switch (exp) - { - case 0: - return std::ldexp(mant, -24); - case 31: - return (mant == 0) - ? std::numeric_limits::infinity() - : std::numeric_limits::quiet_NaN(); - default: - return std::ldexp(mant + 1024, exp - 25); - } - }(); - return sax->number_float((half & 0x8000u) != 0 - ? static_cast(-val) - : static_cast(val), ""); - } + return get_half_float(input_format_t::cbor, false); case 0xFA: // Single-Precision Float (four-byte IEEE 754) { @@ -1539,6 +1414,73 @@ class binary_reader } } + /*! + @brief read a CBOR argument (additional information 24-27) of the width + @ref current announces + + The lower 5 bits of @a current (0x18-0x1B) select a 1/2/4/8-byte + big-endian unsigned integer that follows the head byte; this is shared by + every major type that uses this encoding (unsigned/negative integers, + strings, arrays, maps, tags). Reading always goes through @ref get_number, + so EOF is reported the same way as before this helper existed. + + @param[out] value the decoded argument + @return whether reading succeeded + */ + bool get_cbor_argument(std::uint64_t& value) + { + switch (current & 0x1F) + { + case 0x18: // 1 byte + { + std::uint8_t n{}; + if (JSON_HEDLEY_UNLIKELY(!get_number(input_format_t::cbor, n))) + { + return false; + } + value = n; + return true; + } + + case 0x19: // 2 bytes + { + std::uint16_t n{}; + if (JSON_HEDLEY_UNLIKELY(!get_number(input_format_t::cbor, n))) + { + return false; + } + value = n; + return true; + } + + case 0x1A: // 4 bytes + { + std::uint32_t n{}; + if (JSON_HEDLEY_UNLIKELY(!get_number(input_format_t::cbor, n))) + { + return false; + } + value = n; + return true; + } + + case 0x1B: // 8 bytes + { + std::uint64_t n{}; + if (JSON_HEDLEY_UNLIKELY(!get_number(input_format_t::cbor, n))) + { + return false; + } + value = n; + return true; + } + + default: // LCOV_EXCL_LINE + JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE + return false; // LCOV_EXCL_LINE + } + } + /*! @brief narrow a definite CBOR array/map length to std::size_t @@ -1571,19 +1513,15 @@ class binary_reader enclosing container after each element, so that the nesting depth of the input costs heap rather than native stack (see #5104). - @param[in] get_char whether a new character should be retrieved from the - input (true) or whether the last read character - @a current should be considered instead @param[in] tag_handler how CBOR tags should be treated @return whether reading the value succeeded */ - bool parse_cbor_internal(const bool get_char, - const cbor_tag_handler_t tag_handler) + bool parse_cbor_internal(const cbor_tag_handler_t tag_handler) { // whether the next value starts at a fresh byte or at the one already // read into `current` - bool fetch = get_char; + bool fetch = true; // the key currently being read; hoisted out of the loop so that its // capacity is reused across elements and across nesting levels @@ -1626,8 +1564,7 @@ class binary_reader if (at_end) { - container_stack.pop_back(); - if (JSON_HEDLEY_UNLIKELY(top.is_object ? !sax->end_object() : !sax->end_array())) + if (JSON_HEDLEY_UNLIKELY(!leave_container())) { return false; } @@ -1676,9 +1613,6 @@ class binary_reader // MsgPack // ///////////// - /*! - @return whether a valid MessagePack value was passed to the SAX parser - */ /*! @brief read one MessagePack value @@ -2377,8 +2311,7 @@ class binary_reader if (container_stack.back().remaining == 0) { - container_stack.pop_back(); - if (JSON_HEDLEY_UNLIKELY(is_object ? !sax->end_object() : !sax->end_array())) + if (JSON_HEDLEY_UNLIKELY(!leave_container())) { return false; } @@ -2423,20 +2356,16 @@ class binary_reader //////////// /*! - @param[in] get_char whether a new character should be retrieved from the - input (true, default) or whether the last read - character should be considered instead - @return whether a valid UBJSON value was passed to the SAX parser */ - bool parse_ubjson_internal(const bool get_char = true) + bool parse_ubjson_internal() { // the key currently being read; hoisted out of the loop so that its // capacity is reused across elements and across nesting levels string_t key; // the type marker of the value to read next - char_int_type prefix = get_char ? get_ignore_noop() : current; + char_int_type prefix = get_ignore_noop(); while (true) { @@ -2511,8 +2440,7 @@ class binary_reader break; } - container_stack.pop_back(); - if (JSON_HEDLEY_UNLIKELY(top.is_object ? !sax->end_object() : !sax->end_array())) + if (JSON_HEDLEY_UNLIKELY(!leave_container())) { return false; } @@ -2720,6 +2648,42 @@ class binary_reader return true; } + /*! + @brief read a UBJSON/BJData optimized-container count of a signed marker + type ('i', 'I', 'l', 'L') and narrow it to std::size_t + + Every signed count marker rejects a negative value the same way (error + 113); the value_in_range_of check additionally needed for 'L' is only + ever live when @a SignedType is std::int64_t on a target where + std::size_t is narrower (e.g. 32-bit), since 'i'/'I'/'l' can never exceed + std::size_t there. + + @tparam SignedType std::int8_t, std::int16_t, std::int32_t or std::int64_t + @param[out] result the count narrowed to std::size_t + @return whether reading and validating succeeded + */ + template + bool get_ubjson_signed_count(std::size_t& result) + { + SignedType number{}; + if (JSON_HEDLEY_UNLIKELY(!get_number(input_format, number))) + { + return false; + } + if (JSON_HEDLEY_UNLIKELY(number < 0)) + { + return sax->parse_error(chars_read, get_token_string(), parse_error::create(113, chars_read, + exception_message(input_format, "count in an optimized container must be positive", "size"), nullptr)); + } + if (JSON_HEDLEY_UNLIKELY(!value_in_range_of(number))) + { + return sax->parse_error(chars_read, get_token_string(), out_of_range::create(408, + exception_message(input_format, "integer value overflow", "size"), nullptr)); + } + result = static_cast(number); // NOLINT(bugprone-signed-char-misuse,cert-str34-c): number is not a char + return true; + } + /*! @param[out] result determined size @param[in,out] is_ndarray for input, `true` means already inside an ndarray vector @@ -2752,73 +2716,16 @@ class binary_reader } case 'i': - { - std::int8_t number{}; - if (JSON_HEDLEY_UNLIKELY(!get_number(input_format, number))) - { - return false; - } - if (number < 0) - { - return sax->parse_error(chars_read, get_token_string(), parse_error::create(113, chars_read, - exception_message(input_format, "count in an optimized container must be positive", "size"), nullptr)); - } - result = static_cast(number); // NOLINT(bugprone-signed-char-misuse,cert-str34-c): number is not a char - return true; - } + return get_ubjson_signed_count(result); case 'I': - { - std::int16_t number{}; - if (JSON_HEDLEY_UNLIKELY(!get_number(input_format, number))) - { - return false; - } - if (number < 0) - { - return sax->parse_error(chars_read, get_token_string(), parse_error::create(113, chars_read, - exception_message(input_format, "count in an optimized container must be positive", "size"), nullptr)); - } - result = static_cast(number); - return true; - } + return get_ubjson_signed_count(result); case 'l': - { - std::int32_t number{}; - if (JSON_HEDLEY_UNLIKELY(!get_number(input_format, number))) - { - return false; - } - if (number < 0) - { - return sax->parse_error(chars_read, get_token_string(), parse_error::create(113, chars_read, - exception_message(input_format, "count in an optimized container must be positive", "size"), nullptr)); - } - result = static_cast(number); - return true; - } + return get_ubjson_signed_count(result); case 'L': - { - std::int64_t number{}; - if (JSON_HEDLEY_UNLIKELY(!get_number(input_format, number))) - { - return false; - } - if (number < 0) - { - return sax->parse_error(chars_read, get_token_string(), parse_error::create(113, chars_read, - exception_message(input_format, "count in an optimized container must be positive", "size"), nullptr)); - } - if (!value_in_range_of(number)) - { - return sax->parse_error(chars_read, get_token_string(), out_of_range::create(408, - exception_message(input_format, "integer value overflow", "size"), nullptr)); - } - result = static_cast(number); - return true; - } + return get_ubjson_signed_count(result); case 'u': { @@ -2909,16 +2816,23 @@ class binary_reader result = 1; for (auto i : dim) { - // Pre-multiplication overflow check: if i > 0 and result > SIZE_MAX/i, then result*i would overflow. - // This check must happen before multiplication since overflow detection after the fact is unreliable - // as modular arithmetic can produce any value, not just 0 or SIZE_MAX. - if (JSON_HEDLEY_UNLIKELY(i > 0 && result > (std::numeric_limits::max)() / i)) + // Pre-multiplication overflow check: since the loop above + // already rejected any zero dimension, i is always > 0 + // here, so result > SIZE_MAX/i means result*i would + // overflow. This check must happen before multiplication + // since overflow detection after the fact is unreliable, + // as modular arithmetic can produce any value, not just 0 + // or SIZE_MAX. + if (JSON_HEDLEY_UNLIKELY(result > (std::numeric_limits::max)() / i)) { return sax->parse_error(chars_read, get_token_string(), out_of_range::create(408, exception_message(input_format, "excessive ndarray size caused overflow", "size"), nullptr)); } result *= i; - // Additional post-multiplication check to catch any edge cases the pre-check might miss - if (result == 0 || result == npos) + // the pre-check above already rules out result becoming 0 + // by overflow; the only value it cannot rule out is an + // exact match with npos, the sentinel reserved for an + // unknown-size container (see get_ubjson_size_type()) + if (result == npos) { return sax->parse_error(chars_read, get_token_string(), out_of_range::create(408, exception_message(input_format, "excessive ndarray size caused overflow", "size"), nullptr)); } @@ -2979,7 +2893,7 @@ class binary_reader { result.second = get(); // must not ignore 'N', because 'N' maybe the type if (input_format == input_format_t::bjdata - && JSON_HEDLEY_UNLIKELY(std::binary_search(bjd_optimized_type_markers.begin(), bjd_optimized_type_markers.end(), result.second))) + && JSON_HEDLEY_UNLIKELY(is_bjd_excluded_optimized_type(result.second))) { auto last_token = get_token_string(); return sax->parse_error(chars_read, last_token, parse_error::create(112, chars_read, @@ -3123,50 +3037,7 @@ class binary_reader { break; } - const auto byte1_raw = get(); - if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format, "number"))) - { - return false; - } - const auto byte2_raw = get(); - if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format, "number"))) - { - return false; - } - - const auto byte1 = static_cast(byte1_raw); - const auto byte2 = static_cast(byte2_raw); - - // Code from RFC 8949, Appendix D, Figure 3: - // As half-precision floating-point numbers were only added - // to IEEE 754 in 2008, today's programming platforms often - // still only have limited support for them. It is very - // easy to include at least decoding support for them even - // without such support. An example of a small decoder for - // half-precision floating-point numbers in the C language - // is shown in Fig. 3. - const auto half = static_cast((byte2 << 8u) + byte1); - const double val = [&half] - { - const int exp = (half >> 10u) & 0x1Fu; - const unsigned int mant = half & 0x3FFu; - JSON_ASSERT(exp <= 31); - JSON_ASSERT(mant <= 1023); - switch (exp) - { - case 0: - return std::ldexp(mant, -24); - case 31: - return (mant == 0) - ? std::numeric_limits::infinity() - : std::numeric_limits::quiet_NaN(); - default: - return std::ldexp(mant + 1024, exp - 25); - } - }(); - return sax->number_float((half & 0x8000u) != 0 - ? static_cast(-val) - : static_cast(val), ""); + return get_half_float(input_format, true); } case 'd': @@ -3239,19 +3110,16 @@ class binary_reader if (input_format == input_format_t::bjdata && size_and_type.first != npos && (size_and_type.second & (1 << 8)) != 0) { size_and_type.second &= ~(static_cast(1) << 8); // use bit 8 to indicate ndarray, here we remove the bit to restore the type marker - auto it = std::lower_bound(bjd_types_map.begin(), bjd_types_map.end(), size_and_type.second, [](const bjd_type & p, char_int_type t) - { - return p.first < t; - }); + const char* type_name = bjd_type_name(size_and_type.second); string_t key = "_ArrayType_"; - if (JSON_HEDLEY_UNLIKELY(it == bjd_types_map.end() || it->first != size_and_type.second)) + if (JSON_HEDLEY_UNLIKELY(type_name == nullptr)) { auto last_token = get_token_string(); return sax->parse_error(chars_read, last_token, parse_error::create(112, chars_read, exception_message(input_format, "invalid byte: 0x" + last_token, "type"), nullptr)); } - string_t type = it->second; // sax->string() takes a reference + string_t type = type_name; // sax->string() takes a reference if (JSON_HEDLEY_UNLIKELY(!sax->key(key) || !sax->string(type))) { return false; @@ -3543,8 +3411,7 @@ class binary_reader if (at_end) { - container_stack.pop_back(); - if (JSON_HEDLEY_UNLIKELY(top.is_object ? !sax->end_object() : !sax->end_array())) + if (JSON_HEDLEY_UNLIKELY(!leave_container())) { return false; } @@ -4083,6 +3950,68 @@ class binary_reader return true; } + /*! + @brief read and decode an IEEE 754 half-precision (16-bit) float + + Used by CBOR (big endian) and BJData (little endian); the two formats + only differ in the byte order of the two bytes that make up the half. + + @param[in] format the current format (for diagnostics) + @param[in] little_endian whether the two bytes are little endian (BJData) + or big endian (CBOR) + + @return whether reading and decoding succeeded + */ + bool get_half_float(const input_format_t format, const bool little_endian) + { + const auto byte1_raw = get(); + if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(format, "number"))) + { + return false; + } + const auto byte2_raw = get(); + if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(format, "number"))) + { + return false; + } + + const auto byte1 = static_cast(byte1_raw); + const auto byte2 = static_cast(byte2_raw); + + // Code from RFC 8949, Appendix D, Figure 3: + // As half-precision floating-point numbers were only added + // to IEEE 754 in 2008, today's programming platforms often + // still only have limited support for them. It is very + // easy to include at least decoding support for them even + // without such support. An example of a small decoder for + // half-precision floating-point numbers in the C language + // is shown in Fig. 3. + const auto half = little_endian + ? static_cast((byte2 << 8u) + byte1) + : static_cast((byte1 << 8u) + byte2); + const double val = [&half] + { + const int exp = (half >> 10u) & 0x1Fu; + const unsigned int mant = half & 0x3FFu; + JSON_ASSERT(exp <= 31); + JSON_ASSERT(mant <= 1023); + switch (exp) + { + case 0: + return std::ldexp(mant, -24); + case 31: + return (mant == 0) + ? std::numeric_limits::infinity() + : std::numeric_limits::quiet_NaN(); + default: + return std::ldexp(mant + 1024, exp - 25); + } + }(); + return sax->number_float((half & 0x8000u) != 0 + ? static_cast(-val) + : static_cast(val), ""); + } + /*! @brief create a string by reading characters from the input @@ -4308,38 +4237,61 @@ class binary_reader /// BON8: number of bytes in @ref bon8_pushback std::size_t bon8_pushback_size = 0; - // excluded markers in bjdata optimized type -#define JSON_BINARY_READER_MAKE_BJD_OPTIMIZED_TYPE_MARKERS_ \ - make_array('F', 'H', 'N', 'S', 'T', 'Z', '[', '{') - -#define JSON_BINARY_READER_MAKE_BJD_TYPES_MAP_ \ - make_array( \ - bjd_type{'B', "byte"}, \ - bjd_type{'C', "char"}, \ - bjd_type{'D', "double"}, \ - bjd_type{'I', "int16"}, \ - bjd_type{'L', "int64"}, \ - bjd_type{'M', "uint64"}, \ - bjd_type{'U', "uint8"}, \ - bjd_type{'d', "single"}, \ - bjd_type{'i', "int8"}, \ - bjd_type{'l', "int32"}, \ - bjd_type{'m', "uint32"}, \ - bjd_type{'u', "uint16"}) - JSON_PRIVATE_UNLESS_TESTED: - // lookup tables - // NOLINTNEXTLINE(cppcoreguidelines-non-private-member-variables-in-classes) - const decltype(JSON_BINARY_READER_MAKE_BJD_OPTIMIZED_TYPE_MARKERS_) bjd_optimized_type_markers = - JSON_BINARY_READER_MAKE_BJD_OPTIMIZED_TYPE_MARKERS_; + /*! + @brief whether @a marker is excluded from BJData's optimized ND-array types + @return whether @a marker is one of 'F', 'H', 'N', 'S', 'T', 'Z', '[', '{' - using bjd_type = std::pair; - // NOLINTNEXTLINE(cppcoreguidelines-non-private-member-variables-in-classes) - const decltype(JSON_BINARY_READER_MAKE_BJD_TYPES_MAP_) bjd_types_map = - JSON_BINARY_READER_MAKE_BJD_TYPES_MAP_; + Mirrors binary_writer's @ref binary_writer::is_bjdata_excluded_type_marker + "is_bjdata_excluded_type_marker()`, which encodes the same list the other + way; keep the two in sync. + */ + static constexpr bool is_bjd_excluded_optimized_type(const char_int_type marker) noexcept + { + return marker == '[' || marker == '{' || marker == 'S' || marker == 'H' + || marker == 'T' || marker == 'F' || marker == 'N' || marker == 'Z'; + } -#undef JSON_BINARY_READER_MAKE_BJD_OPTIMIZED_TYPE_MARKERS_ -#undef JSON_BINARY_READER_MAKE_BJD_TYPES_MAP_ + /*! + @brief look up the ND-array element type name for a BJData dtype marker + @return the type name ("uint8", "int8", ...), or nullptr if @a marker does + not name a known dtype + + A C++11 `constexpr` function cannot contain a `switch`, so this is a + plain (non-constexpr) switch instead. + */ + static const char* bjd_type_name(const char_int_type marker) + { + switch (marker) + { + case 'B': + return "byte"; + case 'C': + return "char"; + case 'D': + return "double"; + case 'I': + return "int16"; + case 'L': + return "int64"; + case 'M': + return "uint64"; + case 'U': + return "uint8"; + case 'd': + return "single"; + case 'i': + return "int8"; + case 'l': + return "int32"; + case 'm': + return "uint32"; + case 'u': + return "uint16"; + default: + return nullptr; + } + } }; #ifndef JSON_HAS_CPP_17 diff --git a/include/nlohmann/detail/meta/cpp_future.hpp b/include/nlohmann/detail/meta/cpp_future.hpp index e109feacb..3e012b341 100644 --- a/include/nlohmann/detail/meta/cpp_future.hpp +++ b/include/nlohmann/detail/meta/cpp_future.hpp @@ -9,7 +9,6 @@ #pragma once -#include // array #include // size_t #include // conditional, enable_if, false_type, integral_constant, is_constructible, is_integral, is_same, remove_cv, remove_reference, true_type #include // index_sequence, make_index_sequence, index_sequence_for @@ -161,11 +160,5 @@ struct static_const constexpr T static_const::value; #endif -template -constexpr std::array make_array(Args&& ... args) -{ - return std::array {{static_cast(std::forward(args))...}}; -} - } // namespace detail NLOHMANN_JSON_NAMESPACE_END diff --git a/include/nlohmann/detail/output/binary_writer.hpp b/include/nlohmann/detail/output/binary_writer.hpp index 1ebd3c8ab..f20379436 100644 --- a/include/nlohmann/detail/output/binary_writer.hpp +++ b/include/nlohmann/detail/output/binary_writer.hpp @@ -10,7 +10,6 @@ #include // reverse #include // array -#include // map #include // isnan, isinf #include // uint8_t, uint16_t, uint32_t, uint64_t #include // memcpy @@ -115,7 +114,7 @@ class binary_writer /*! @param[in] j JSON value to serialize - @pre j.type() == value_t::object + @throw type_error.317 if @a j is not an object */ void write_bson(const BasicJsonType& j) { @@ -204,7 +203,7 @@ class binary_writer } else { - write_compact_float(j.m_data.m_value.number_float, detail::input_format_t::cbor); + write_compact_float(j.m_data.m_value.number_float, to_char_type(0xFA), to_char_type(0xFB)); } break; } @@ -238,6 +237,13 @@ class binary_writer { if (j.m_data.m_value.binary->has_subtype()) { + // The subtype is always written as a tag with a 0xD8..0xDB + // head, never in the one-byte form 0xC0..0xD7 that CBOR + // allows for tags 0..23 (so this is not write_cbor_head). + // binary_reader with cbor_tag_handler_t::store only turns + // 0xD8..0xDB into a subtype and ignores the one-byte tags, + // so the shorter form would lose subtypes 0..23 on a round + // trip. if (j.m_data.m_value.binary->subtype() <= (std::numeric_limits::max)()) { write_number(static_cast(0xd8)); @@ -309,6 +315,43 @@ class binary_writer return static_cast(length); } + /*! + @brief write a non-negative integer using the MessagePack fixint/uint ladder + @param[in] n the value to write, already known to be non-negative + */ + void write_msgpack_unsigned(const std::uint64_t n) + { + if (n < 128) + { + // positive fixnum + write_number(static_cast(n)); + } + else if (n <= (std::numeric_limits::max)()) + { + // uint 8 + oa.write_character(to_char_type(0xCC)); + write_number(static_cast(n)); + } + else if (n <= (std::numeric_limits::max)()) + { + // uint 16 + oa.write_character(to_char_type(0xCD)); + write_number(static_cast(n)); + } + else if (n <= (std::numeric_limits::max)()) + { + // uint 32 + oa.write_character(to_char_type(0xCE)); + write_number(static_cast(n)); + } + else + { + // uint 64 + oa.write_character(to_char_type(0xCF)); + write_number(n); + } + } + /*! @param[in] j JSON value to serialize */ @@ -335,38 +378,8 @@ class binary_writer if (j.m_data.m_value.number_integer >= 0) { // MessagePack does not differentiate between positive - // signed integers and unsigned integers. Therefore, we used - // the code from the value_t::number_unsigned case here. - const auto value_as_unsigned = static_cast(j.m_data.m_value.number_integer); - if (value_as_unsigned < 128) - { - // positive fixnum - write_number(static_cast(j.m_data.m_value.number_integer)); - } - else if (value_as_unsigned <= (std::numeric_limits::max)()) - { - // uint 8 - oa.write_character(to_char_type(0xCC)); - write_number(static_cast(j.m_data.m_value.number_integer)); - } - else if (value_as_unsigned <= (std::numeric_limits::max)()) - { - // uint 16 - oa.write_character(to_char_type(0xCD)); - write_number(static_cast(j.m_data.m_value.number_integer)); - } - else if (value_as_unsigned <= (std::numeric_limits::max)()) - { - // uint 32 - oa.write_character(to_char_type(0xCE)); - write_number(static_cast(j.m_data.m_value.number_integer)); - } - else - { - // uint 64 - oa.write_character(to_char_type(0xCF)); - write_number(static_cast(j.m_data.m_value.number_integer)); - } + // signed integers and unsigned integers. + write_msgpack_unsigned(static_cast(j.m_data.m_value.number_integer)); } else { @@ -408,41 +421,13 @@ class binary_writer case value_t::number_unsigned: { - if (j.m_data.m_value.number_unsigned < 128) - { - // positive fixnum - write_number(static_cast(j.m_data.m_value.number_unsigned)); - } - else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) - { - // uint 8 - oa.write_character(to_char_type(0xCC)); - write_number(static_cast(j.m_data.m_value.number_unsigned)); - } - else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) - { - // uint 16 - oa.write_character(to_char_type(0xCD)); - write_number(static_cast(j.m_data.m_value.number_unsigned)); - } - else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) - { - // uint 32 - oa.write_character(to_char_type(0xCE)); - write_number(static_cast(j.m_data.m_value.number_unsigned)); - } - else - { - // uint 64 - oa.write_character(to_char_type(0xCF)); - write_number(static_cast(j.m_data.m_value.number_unsigned)); - } + write_msgpack_unsigned(static_cast(j.m_data.m_value.number_unsigned)); break; } case value_t::number_float: { - write_compact_float(j.m_data.m_value.number_float, detail::input_format_t::msgpack); + write_compact_float(j.m_data.m_value.number_float, to_char_type(0xCA), to_char_type(0xCB)); break; } @@ -1383,6 +1368,14 @@ class binary_writer oa.write_character(to_char_type(0x00)); if (parents.empty()) { + // calc_bson_sizes() and write_bson_document() are two + // hand-synchronized passes over the same structure, linked + // only by nested_sizes' visiting order; this checks that the + // write pass consumed exactly the sizes the size pass + // produced, so a future change that desyncs them (skips or + // rejects an entry in only one pass) is caught immediately + // instead of silently writing wrong length prefixes. + JSON_ASSERT(next_size == nested_sizes.size()); return; } current = std::move(parents.back()); @@ -1434,52 +1427,6 @@ class binary_writer } } - static constexpr CharType get_cbor_float_prefix(float /*unused*/) - { - return to_char_type(0xFA); // Single-Precision Float - } - - static constexpr CharType get_cbor_float_prefix(double /*unused*/) - { - return to_char_type(0xFB); // Double-Precision Float - } - - ///////////// - // MsgPack // - ///////////// - - static constexpr CharType get_msgpack_float_prefix(float /*unused*/) - { - return to_char_type(0xCA); // float 32 - } - - static constexpr CharType get_msgpack_float_prefix(double /*unused*/) - { - return to_char_type(0xCB); // float 64 - } - - /// @return the BON8 type marker for binary32 (float) or binary64 (double) - template - static constexpr CharType get_bon8_float_prefix() - { - return to_char_type(std::is_same::value ? 0x8E : 0x8F); - } - - /// @return the type marker for a FloatType value in @a format (CBOR, MessagePack, or BON8) - template - static CharType get_compact_float_prefix(const detail::input_format_t format) - { - if (format == detail::input_format_t::cbor) - { - return get_cbor_float_prefix(FloatType{}); - } - if (format == detail::input_format_t::bon8) - { - return get_bon8_float_prefix(); - } - return get_msgpack_float_prefix(FloatType{}); - } - //////////// // UBJSON // //////////// @@ -1493,207 +1440,127 @@ class binary_writer { if (add_prefix) { - oa.write_character(get_ubjson_float_prefix(n)); + oa.write_character(get_ubjson_float_prefix()); } write_number(n, use_bjdata); } - // UBJSON: write number (unsigned integer) + // UBJSON: write number (integer) template::value, int>::type = 0> + std::is_integral::value, int>::type = 0> void write_number_with_ubjson_prefix(const NumberType n, const bool add_prefix, const bool use_bjdata) { - if (n <= static_cast((std::numeric_limits::max)())) - { - if (add_prefix) - { - oa.write_character(to_char_type('i')); // int8 - } - write_number(static_cast(n), use_bjdata); - } - else if (n <= (std::numeric_limits::max)()) - { - if (add_prefix) - { - oa.write_character(to_char_type('U')); // uint8 - } - write_number(static_cast(n), use_bjdata); - } - else if (n <= static_cast((std::numeric_limits::max)())) - { - if (add_prefix) - { - oa.write_character(to_char_type('I')); // int16 - } - write_number(static_cast(n), use_bjdata); - } - else if (use_bjdata && n <= static_cast((std::numeric_limits::max)())) - { - if (add_prefix) - { - oa.write_character(to_char_type('u')); // uint16 - bjdata only - } - write_number(static_cast(n), use_bjdata); - } - else if (n <= static_cast((std::numeric_limits::max)())) - { - if (add_prefix) - { - oa.write_character(to_char_type('l')); // int32 - } - write_number(static_cast(n), use_bjdata); - } - else if (use_bjdata && n <= static_cast((std::numeric_limits::max)())) - { - if (add_prefix) - { - oa.write_character(to_char_type('m')); // uint32 - bjdata only - } - write_number(static_cast(n), use_bjdata); - } - else if (n <= static_cast((std::numeric_limits::max)())) - { - if (add_prefix) - { - oa.write_character(to_char_type('L')); // int64 - } - write_number(static_cast(n), use_bjdata); - } - else if (use_bjdata) - { - if (add_prefix) - { - oa.write_character(to_char_type('M')); // uint64 - bjdata only - } - write_number(static_cast(n), use_bjdata); - } - else - { - if (add_prefix) - { - oa.write_character(to_char_type('H')); // high-precision number - } - - const auto number = BasicJsonType(n).dump(); - write_number_with_ubjson_prefix(number.size(), true, use_bjdata); - for (std::size_t i = 0; i < number.size(); ++i) - { - oa.write_character(to_char_type(static_cast(number[i]))); - } - } - } - - // UBJSON: write number (signed integer) - template < typename NumberType, typename std::enable_if < - std::is_signed::value&& - !std::is_floating_point::value, int >::type = 0 > - void write_number_with_ubjson_prefix(const NumberType n, - const bool add_prefix, - const bool use_bjdata) - { - if ((std::numeric_limits::min)() <= n && n <= (std::numeric_limits::max)()) - { - if (add_prefix) - { - oa.write_character(to_char_type('i')); // int8 - } - write_number(static_cast(n), use_bjdata); - } - else if (static_cast((std::numeric_limits::min)()) <= n && n <= static_cast((std::numeric_limits::max)())) - { - if (add_prefix) - { - oa.write_character(to_char_type('U')); // uint8 - } - write_number(static_cast(n), use_bjdata); - } - else if ((std::numeric_limits::min)() <= n && n <= (std::numeric_limits::max)()) - { - if (add_prefix) - { - oa.write_character(to_char_type('I')); // int16 - } - write_number(static_cast(n), use_bjdata); - } - else if (use_bjdata && (static_cast((std::numeric_limits::min)()) <= n && n <= static_cast((std::numeric_limits::max)()))) - { - if (add_prefix) - { - oa.write_character(to_char_type('u')); // uint16 - bjdata only - } - write_number(static_cast(n), use_bjdata); - } - else if ((std::numeric_limits::min)() <= n && n <= (std::numeric_limits::max)()) - { - if (add_prefix) - { - oa.write_character(to_char_type('l')); // int32 - } - write_number(static_cast(n), use_bjdata); - } - else if (use_bjdata && (static_cast((std::numeric_limits::min)()) <= n && n <= static_cast((std::numeric_limits::max)()))) - { - if (add_prefix) - { - oa.write_character(to_char_type('m')); // uint32 - bjdata only - } - write_number(static_cast(n), use_bjdata); - } - else - { - // every value of an integer type of at most 64 bits fits into an - // int64; only a wider type needs a range check - write_ubjson_int64_or_high_precision(n, add_prefix, use_bjdata, - std::integral_constant < bool, std::numeric_limits::digits <= std::numeric_limits::digits > {}); - } - } - - template - void write_ubjson_int64_or_high_precision(const NumberType n, const bool add_prefix, const bool use_bjdata, std::true_type /*fits_int64*/) - { + const CharType prefix = ubjson_integer_prefix(n, use_bjdata); if (add_prefix) { - oa.write_character(to_char_type('L')); // int64 + oa.write_character(prefix); } - write_number(static_cast(n), use_bjdata); + write_ubjson_integer_payload(prefix, n, use_bjdata); } + /*! + @brief determine the UBJSON/BJData type marker of an integer + + This is the only place that picks the marker of an integer: both + write_number_with_ubjson_prefix() and ubjson_prefix() use it. An optimized + container announces the marker of its first value after `$` and then + writes every value without a marker, so the two must never disagree. + + @param[in] n the integer + @param[in] use_bjdata whether the BJData-only markers `u`, `m`, and `M` + may be used + + @return the first marker of `i`, `U`, `I`, `u` (BJData), `l`, `m` (BJData), + `L`, `M` (BJData, unsigned types only), and `H` (high-precision + number) whose range contains @a n + */ template - void write_ubjson_int64_or_high_precision(const NumberType n, const bool add_prefix, const bool use_bjdata, std::false_type /*fits_int64*/) + static CharType ubjson_integer_prefix(const NumberType n, const bool use_bjdata) noexcept { - if ((std::numeric_limits::min)() <= n && n <= (std::numeric_limits::max)()) + if (value_in_range_of(n)) { - write_ubjson_int64_or_high_precision(n, add_prefix, use_bjdata, std::true_type {}); - return; + return 'i'; } - - if (add_prefix) + if (value_in_range_of(n)) { - oa.write_character(to_char_type('H')); // high-precision number + return 'U'; } - - const auto number = BasicJsonType(n).dump(); - write_number_with_ubjson_prefix(number.size(), true, use_bjdata); - for (std::size_t i = 0; i < number.size(); ++i) + if (value_in_range_of(n)) { - oa.write_character(to_char_type(static_cast(number[i]))); + return 'I'; } + if (use_bjdata && value_in_range_of(n)) + { + return 'u'; + } + if (value_in_range_of(n)) + { + return 'l'; + } + if (use_bjdata && value_in_range_of(n)) + { + return 'm'; + } + if (value_in_range_of(n)) + { + return 'L'; + } + if (use_bjdata && std::is_unsigned::value) + { + return 'M'; + } + // anything else is treated as a high-precision number + return 'H'; } + /*! + @brief write the value of an integer for the marker chosen by + ubjson_integer_prefix() + */ template - static constexpr CharType ubjson_int64_or_high_precision_prefix(const NumberType /*n*/, std::true_type /*fits_int64*/) noexcept + void write_ubjson_integer_payload(const CharType prefix, const NumberType n, const bool use_bjdata) { - return 'L'; - } - - template - static CharType ubjson_int64_or_high_precision_prefix(const NumberType n, std::false_type /*fits_int64*/) noexcept - { - // anything outside of the range of an int64 is treated as a - // high-precision number - return ((std::numeric_limits::min)() <= n && n <= (std::numeric_limits::max)()) ? 'L' : 'H'; + switch (prefix) + { + case 'i': + write_number(static_cast(n), use_bjdata); + break; + case 'U': + write_number(static_cast(n), use_bjdata); + break; + case 'I': + write_number(static_cast(n), use_bjdata); + break; + case 'u': + write_number(static_cast(n), use_bjdata); + break; + case 'l': + write_number(static_cast(n), use_bjdata); + break; + case 'm': + write_number(static_cast(n), use_bjdata); + break; + case 'L': + write_number(static_cast(n), use_bjdata); + break; + case 'M': + write_number(static_cast(n), use_bjdata); + break; + default: + { + // high-precision number: the decimal digits as a string + JSON_ASSERT(prefix == 'H'); + const auto number = BasicJsonType(n).dump(); + write_number_with_ubjson_prefix(number.size(), true, use_bjdata); + for (std::size_t i = 0; i < number.size(); ++i) + { + oa.write_character(to_char_type(static_cast(number[i]))); + } + break; + } + } } /*! @@ -1710,77 +1577,13 @@ class binary_writer return j.m_data.m_value.boolean ? 'T' : 'F'; case value_t::number_integer: - { - if ((std::numeric_limits::min)() <= j.m_data.m_value.number_integer && j.m_data.m_value.number_integer <= (std::numeric_limits::max)()) - { - return 'i'; - } - if ((std::numeric_limits::min)() <= j.m_data.m_value.number_integer && j.m_data.m_value.number_integer <= (std::numeric_limits::max)()) - { - return 'U'; - } - if ((std::numeric_limits::min)() <= j.m_data.m_value.number_integer && j.m_data.m_value.number_integer <= (std::numeric_limits::max)()) - { - return 'I'; - } - if (use_bjdata && ((std::numeric_limits::min)() <= j.m_data.m_value.number_integer && j.m_data.m_value.number_integer <= (std::numeric_limits::max)())) - { - return 'u'; - } - if ((std::numeric_limits::min)() <= j.m_data.m_value.number_integer && j.m_data.m_value.number_integer <= (std::numeric_limits::max)()) - { - return 'l'; - } - if (use_bjdata && ((std::numeric_limits::min)() <= j.m_data.m_value.number_integer && j.m_data.m_value.number_integer <= (std::numeric_limits::max)())) - { - return 'm'; - } - // every value of an integer type of at most 64 bits fits into - // an int64; only a wider type needs a range check - return ubjson_int64_or_high_precision_prefix(j.m_data.m_value.number_integer, - std::integral_constant < bool, std::numeric_limits::digits <= std::numeric_limits::digits > {}); - } + return ubjson_integer_prefix(j.m_data.m_value.number_integer, use_bjdata); case value_t::number_unsigned: - { - if (j.m_data.m_value.number_unsigned <= static_cast((std::numeric_limits::max)())) - { - return 'i'; - } - if (j.m_data.m_value.number_unsigned <= static_cast((std::numeric_limits::max)())) - { - return 'U'; - } - if (j.m_data.m_value.number_unsigned <= static_cast((std::numeric_limits::max)())) - { - return 'I'; - } - if (use_bjdata && j.m_data.m_value.number_unsigned <= static_cast((std::numeric_limits::max)())) - { - return 'u'; - } - if (j.m_data.m_value.number_unsigned <= static_cast((std::numeric_limits::max)())) - { - return 'l'; - } - if (use_bjdata && j.m_data.m_value.number_unsigned <= static_cast((std::numeric_limits::max)())) - { - return 'm'; - } - if (j.m_data.m_value.number_unsigned <= static_cast((std::numeric_limits::max)())) - { - return 'L'; - } - if (use_bjdata) - { - return 'M'; - } - // anything else is treated as a high-precision number - return 'H'; - } + return ubjson_integer_prefix(j.m_data.m_value.number_unsigned, use_bjdata); case value_t::number_float: - return get_ubjson_float_prefix(j.m_data.m_value.number_float); + return get_ubjson_float_prefix(); case value_t::string: return 'S'; @@ -1805,7 +1608,7 @@ class binary_writer Containers, strings, high-precision numbers, booleans and null cannot be declared as the single type of an optimized container in BJData; such a container is written unoptimized. The reader rejects them with the same - list (binary_reader::bjd_optimized_type_markers). + list (binary_reader::is_bjd_excluded_optimized_type()). */ static constexpr bool is_bjdata_excluded_type_marker(const CharType marker) noexcept { @@ -1813,14 +1616,16 @@ class binary_writer || marker == 'T' || marker == 'F' || marker == 'N' || marker == 'Z'; } - static constexpr CharType get_ubjson_float_prefix(float /*unused*/) + /// @return the UBJSON/BJData type marker for a float or double value + /// + /// number_float_t must be float or double; a static_assert (rather than + /// an ambiguous overload) reports an unsupported number_float_t clearly. + template + static constexpr CharType get_ubjson_float_prefix() { - return 'd'; // float 32 - } - - static constexpr CharType get_ubjson_float_prefix(double /*unused*/) - { - return 'D'; // float 64 + static_assert(std::is_same::value || std::is_same::value, + "number_float_t must be float or double for the UBJSON/BJData writer"); + return std::is_same::value ? 'd' : 'D'; // float 32 / float 64 } /*! @@ -1837,35 +1642,184 @@ class binary_writer : value_in_range_of(el.template get()); } + /*! + @brief look up the BJData ND-array dtype marker for an `_ArrayType_` name + @return the one-character marker, or '\0' if @a name does not name a known dtype + + A C++11 `constexpr` function cannot contain a `switch`, so this is a plain + comparison chain instead; it is only reached once per ND-array candidate + object. Keep in sync with binary_reader's `bjd_type_name()`, which maps + the other way. + */ + static CharType bjdata_ndarray_type_marker(const string_t& name) + { + if (name == "uint8") + { + return 'U'; + } + if (name == "int8") + { + return 'i'; + } + if (name == "uint16") + { + return 'u'; + } + if (name == "int16") + { + return 'I'; + } + if (name == "uint32") + { + return 'm'; + } + if (name == "int32") + { + return 'l'; + } + if (name == "uint64") + { + return 'M'; + } + if (name == "int64") + { + return 'L'; + } + if (name == "single") + { + return 'd'; + } + if (name == "double") + { + return 'D'; + } + if (name == "char") + { + return 'C'; + } + if (name == "byte") + { + return 'B'; + } + return '\0'; + } + + /*! + @brief validate (dry_run) or write one BJData ND-array element of integer dtype @a T + @return whether @a el's value is in range of @a T; always true when @a dry_run is false + */ + template + bool write_bjdata_ndarray_element(const BasicJsonType& el, const bool dry_run) + { + if (dry_run) + { + return bjdata_ndarray_value_in_range(el); + } + using storage_type = typename std::conditional::value, std::uint64_t, std::int64_t>::type; + write_number(static_cast(el.template get()), true); + return true; + } + + /*! + @brief validate (dry_run) or write one BJData ND-array element of dtype 'd' (single precision) + @return whether @a el's value fits a float without overflow; always true when @a dry_run is false + */ + bool write_bjdata_ndarray_float_element(const BasicJsonType& el, const bool dry_run) + { + const auto dval = el.template get(); + if (dry_run) + { + return !std::isfinite(dval) || + (dval >= static_cast(std::numeric_limits::lowest()) && + dval <= static_cast((std::numeric_limits::max)())); + } + write_number(static_cast(dval), true); + return true; + } + + /*! + @brief validate or write every element of a BJData ND-array's `_ArrayData_` + @param[in] array_data the `_ArrayData_` array + @param[in] dtype the ND-array dtype marker, as returned by bjdata_ndarray_type_marker() + @param[in] dry_run true to only range-check each element, false to write it + @return whether every element is in range for @a dtype (always true when @a dry_run is false) + */ + bool write_bjdata_ndarray_elements(const BasicJsonType& array_data, const CharType dtype, const bool dry_run) + { + for (const auto& el : array_data) + { + bool ok = true; + switch (dtype) + { + case 'U': + case 'C': + case 'B': + ok = write_bjdata_ndarray_element(el, dry_run); + break; + case 'i': + ok = write_bjdata_ndarray_element(el, dry_run); + break; + case 'u': + ok = write_bjdata_ndarray_element(el, dry_run); + break; + case 'I': + ok = write_bjdata_ndarray_element(el, dry_run); + break; + case 'm': + ok = write_bjdata_ndarray_element(el, dry_run); + break; + case 'l': + ok = write_bjdata_ndarray_element(el, dry_run); + break; + case 'M': + ok = write_bjdata_ndarray_element(el, dry_run); + break; + case 'L': + ok = write_bjdata_ndarray_element(el, dry_run); + break; + case 'd': + ok = write_bjdata_ndarray_float_element(el, dry_run); + break; + case 'D': + default: + // 'D' (double) already spans the full range of number_float_t + if (!dry_run) + { + write_number(el.template get(), true); + } + break; + } + if (!ok) + { + return false; + } + } + return true; + } + /*! @return false if the object is successfully converted to a bjdata ndarray, true if the type or size is invalid */ bool write_bjdata_ndarray(const typename BasicJsonType::object_t& value, const bool use_count, const bool use_type, const bjdata_version_t bjdata_version) { - std::map bjdtype = {{"uint8", 'U'}, {"int8", 'i'}, {"uint16", 'u'}, {"int16", 'I'}, - {"uint32", 'm'}, {"int32", 'l'}, {"uint64", 'M'}, {"int64", 'L'}, {"single", 'd'}, {"double", 'D'}, - {"char", 'C'}, {"byte", 'B'} - }; - - string_t key = "_ArrayType_"; + const auto& array_type = value.at("_ArrayType_"); // the type name is looked up as a string below; a non-string // annotation (e.g. a number, null, or an array) cannot name a known // dtype, so it is treated the same as an unrecognized type name and // falls back to a plain object encoding instead of throwing // type_error.302 out of get() - if (!value.at(key).is_string()) + if (!array_type.is_string()) { return true; } // use get() instead of static_cast to avoid an // ambiguous conversion under explicit instantiation on C++17 (see #4825) - auto it = bjdtype.find(value.at(key).template get()); - if (it == bjdtype.end()) + const CharType dtype = bjdata_ndarray_type_marker(array_type.template get()); + if (dtype == '\0') { return true; } - CharType dtype = it->second; // the 'B' (byte) marker is only defined from BJData Draft 3 onward; // emitting it under an earlier draft would produce a stream that an @@ -1877,12 +1831,12 @@ class binary_writer return true; } - key = "_ArraySize_"; + const auto& array_size = value.at("_ArraySize_"); // the dimensions are written verbatim as the header length below, so a // value that is not an array cannot produce a valid one: null emits 'Z' // and an object emits '{', neither of which a reader accepts after '#'. // Such an object is not a valid ndarray and falls back to a plain object. - if (!value.at(key).is_array()) + if (!array_size.is_array()) { return true; } @@ -1892,7 +1846,7 @@ class binary_writer // dimension, or a 1xN row vector is read back as a plain array, which // would silently drop the annotation, so such an object falls back to // a plain object encoding instead - const auto& dims = value.at(key); + const auto& dims = array_size; if (dims.size() < 2 || (dims.size() == 2 && dims.at(0).is_number_integer() && dims.at(0).template get() == 1)) { return true; @@ -1939,8 +1893,8 @@ class binary_writer // to be an array: size() is 0 for null and 1 for any other scalar, and // iterating an object visits its values, so any of these could match // the dimensions by accident and be encoded as an unrelated ND-array - key = "_ArrayData_"; - if (!value.at(key).is_array() || value.at(key).size() != len) + const auto& array_data = value.at("_ArrayData_"); + if (!array_data.is_array() || array_data.size() != len) { return true; } @@ -1955,7 +1909,7 @@ class binary_writer // API stores int literals as signed), so both are accepted here and the // writes below go through get<>, which reads the member that is active. const bool ndarray_is_float = (dtype == 'd' || dtype == 'D'); - for (const auto& el : value.at(key)) + for (const auto& el : array_data) { if (ndarray_is_float ? !el.is_number_float() : !el.is_number_integer()) { @@ -1968,134 +1922,19 @@ class binary_writer // wrap (integers) or overflow to infinity (the "single" precision // float) instead of being reported, so such an object falls back to // a plain object encoding as well - for (const auto& el : value.at(key)) + if (!write_bjdata_ndarray_elements(array_data, dtype, true)) { - bool in_range = true; - switch (dtype) - { - case 'U': - case 'C': - case 'B': - in_range = bjdata_ndarray_value_in_range(el); - break; - case 'i': - in_range = bjdata_ndarray_value_in_range(el); - break; - case 'u': - in_range = bjdata_ndarray_value_in_range(el); - break; - case 'I': - in_range = bjdata_ndarray_value_in_range(el); - break; - case 'm': - in_range = bjdata_ndarray_value_in_range(el); - break; - case 'l': - in_range = bjdata_ndarray_value_in_range(el); - break; - case 'M': - in_range = bjdata_ndarray_value_in_range(el); - break; - case 'L': - in_range = bjdata_ndarray_value_in_range(el); - break; - case 'd': - { - const auto dval = el.template get(); - in_range = !std::isfinite(dval) || - (dval >= static_cast(std::numeric_limits::lowest()) && - dval <= static_cast((std::numeric_limits::max)())); - break; - } - default: - // 'D' (double) already spans the full range of number_float_t - break; - } - if (!in_range) - { - return true; - } + return true; } - oa.write_character('['); - oa.write_character('$'); + oa.write_character(to_char_type('[')); + oa.write_character(to_char_type('$')); oa.write_character(dtype); - oa.write_character('#'); + oa.write_character(to_char_type('#')); - key = "_ArraySize_"; - write_ubjson(value.at(key), use_count, use_type, true, true, bjdata_version); + write_ubjson(array_size, use_count, use_type, true, true, bjdata_version); - key = "_ArrayData_"; - if (dtype == 'U' || dtype == 'C' || dtype == 'B') - { - for (const auto& el : value.at(key)) - { - write_number(static_cast(el.template get()), true); - } - } - else if (dtype == 'i') - { - for (const auto& el : value.at(key)) - { - write_number(static_cast(el.template get()), true); - } - } - else if (dtype == 'u') - { - for (const auto& el : value.at(key)) - { - write_number(static_cast(el.template get()), true); - } - } - else if (dtype == 'I') - { - for (const auto& el : value.at(key)) - { - write_number(static_cast(el.template get()), true); - } - } - else if (dtype == 'm') - { - for (const auto& el : value.at(key)) - { - write_number(static_cast(el.template get()), true); - } - } - else if (dtype == 'l') - { - for (const auto& el : value.at(key)) - { - write_number(static_cast(el.template get()), true); - } - } - else if (dtype == 'M') - { - for (const auto& el : value.at(key)) - { - write_number(el.template get(), true); - } - } - else if (dtype == 'L') - { - for (const auto& el : value.at(key)) - { - write_number(el.template get(), true); - } - } - else if (dtype == 'd') - { - for (const auto& el : value.at(key)) - { - write_number(static_cast(el.template get()), true); - } - } - else if (dtype == 'D') - { - for (const auto& el : value.at(key)) - { - write_number(el.template get(), true); - } - } + write_bjdata_ndarray_elements(array_data, dtype, false); return false; } @@ -2408,7 +2247,7 @@ class binary_writer } else { - write_compact_float(n, detail::input_format_t::bon8); + write_compact_float(n, to_char_type(0x8E), to_char_type(0x8F)); } #ifdef __GNUC__ JSON_HEDLEY_DIAGNOSTIC_POP @@ -2419,19 +2258,6 @@ class binary_writer // Utility functions // /////////////////////// - /* - @brief write a number to output input - @param[in] n number of type @a NumberType - @param[in] OutputIsLittleEndian Set to true if output data is - required to be little endian - @tparam NumberType the type of the number - - @note This function needs to respect the system's endianness, because bytes - in CBOR, MessagePack, and UBJSON are stored in network order (big - endian) and therefore need reordering on little endian systems. - On the other hand, BSON and BJData use little endian and should reorder - on big endian systems. - */ // single-instruction byte swaps (compilers lower these to bswap/rev/movbe); // used to emit big-endian numbers without a per-byte std::reverse loop static std::uint16_t byte_swap(std::uint16_t x) noexcept @@ -2513,6 +2339,19 @@ class binary_writer std::reverse(a.begin(), a.end()); } + /*! + @brief write a number to the output + @param[in] n number of type @a NumberType + @param[in] OutputIsLittleEndian Set to true if output data is + required to be little endian + @tparam NumberType the type of the number + + @note This function needs to respect the system's endianness, because bytes + in CBOR, MessagePack, UBJSON, and BON8 are stored in network order + (big endian) and therefore need reordering on little endian systems. + On the other hand, BSON and BJData use little endian and should + reorder on big endian systems. + */ template void write_number(const NumberType n, const bool OutputIsLittleEndian = false) { @@ -2530,8 +2369,17 @@ class binary_writer oa.write_characters(vec.data(), sizeof(NumberType)); } - void write_compact_float(const number_float_t n, detail::input_format_t format) + /// @brief write @a n using @a float32_marker if it round-trips through + /// float, otherwise using @a float64_marker + /// + /// @a float32_marker and @a float64_marker are the format-specific type + /// markers (CBOR: 0xFA/0xFB, MessagePack: 0xCA/0xCB, BON8: 0x8E/0x8F); + /// each caller already knows them at compile time, so the format itself + /// no longer needs to be passed in. + void write_compact_float(const number_float_t n, const CharType float32_marker, const CharType float64_marker) { + static_assert(std::is_same::value || std::is_same::value, + "number_float_t must be float or double for the CBOR/MessagePack/BON8 writer"); #ifdef __GNUC__ JSON_HEDLEY_DIAGNOSTIC_PUSH JSON_HEDLEY_PRAGMA(GCC diagnostic ignored "-Wfloat-equal") @@ -2549,12 +2397,12 @@ class binary_writer static_cast(n) <= static_cast((std::numeric_limits::max)()) && static_cast(static_cast(n)) == static_cast(n)))) { - oa.write_character(get_compact_float_prefix(format)); + oa.write_character(float32_marker); write_number(static_cast(n)); } else { - oa.write_character(get_compact_float_prefix(format)); + oa.write_character(float64_marker); write_number(n); } #ifdef __GNUC__ @@ -2563,7 +2411,7 @@ class binary_writer } public: - // The following to_char_type functions are implement the conversion + // The following to_char_type functions implement the conversion // between uint8_t and CharType. In case CharType is not unsigned, // such a conversion is required to allow values greater than 128. // See for a discussion. diff --git a/include/nlohmann/json.hpp b/include/nlohmann/json.hpp index 344c9209b..09a1ec9b5 100644 --- a/include/nlohmann/json.hpp +++ b/include/nlohmann/json.hpp @@ -5219,7 +5219,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec auto ia = detail::input_adapter(std::forward(i)); return format == input_format_t::json ? parser(std::move(ia), nullptr, true, ignore_comments, ignore_trailing_commas).sax_parse(sax, strict) - : detail::binary_reader(std::move(ia), format).sax_parse(format, sax, strict); + : detail::binary_reader(std::move(ia), format).sax_parse(sax, strict); } /// @brief generate SAX events (iterator pair, or iterator+sentinel pair for C++20 ranges support) @@ -5236,7 +5236,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec auto ia = detail::input_adapter(std::move(first), std::move(last)); return format == input_format_t::json ? parser(std::move(ia), nullptr, true, ignore_comments, ignore_trailing_commas).sax_parse(sax, strict) - : detail::binary_reader(std::move(ia), format).sax_parse(format, sax, strict); + : detail::binary_reader(std::move(ia), format).sax_parse(sax, strict); } /// @brief generate SAX events @@ -5258,7 +5258,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg) ? parser(std::move(ia), nullptr, true, ignore_comments, ignore_trailing_commas).sax_parse(sax, strict) // NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg) - : detail::binary_reader(std::move(ia), format).sax_parse(format, sax, strict); + : detail::binary_reader(std::move(ia), format).sax_parse(sax, strict); } #ifndef JSON_NO_IO /// @brief deserialize from stream @@ -5556,7 +5556,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::forward(i)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::cbor).sax_parse(input_format_t::cbor, &sdp, strict, tag_handler)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::cbor).sax_parse(&sdp, strict, tag_handler)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } @@ -5576,7 +5576,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::move(first), std::move(last)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::cbor).sax_parse(input_format_t::cbor, &sdp, strict, tag_handler)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::cbor).sax_parse(&sdp, strict, tag_handler)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } @@ -5605,7 +5605,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec auto ia = i.get(); detail::json_sax_dom_parser sdp(result, allow_exceptions); // NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg) - if (!binary_reader(std::move(ia), input_format_t::cbor).sax_parse(input_format_t::cbor, &sdp, strict, tag_handler)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::cbor).sax_parse(&sdp, strict, tag_handler)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } @@ -5623,7 +5623,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::forward(i)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::msgpack).sax_parse(input_format_t::msgpack, &sdp, strict)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::msgpack).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } @@ -5642,7 +5642,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::move(first), std::move(last)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::msgpack).sax_parse(input_format_t::msgpack, &sdp, strict)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::msgpack).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } @@ -5669,7 +5669,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec auto ia = i.get(); detail::json_sax_dom_parser sdp(result, allow_exceptions); // NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg) - if (!binary_reader(std::move(ia), input_format_t::msgpack).sax_parse(input_format_t::msgpack, &sdp, strict)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::msgpack).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } @@ -5687,7 +5687,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::forward(i)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::ubjson).sax_parse(input_format_t::ubjson, &sdp, strict)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::ubjson).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } @@ -5706,7 +5706,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::move(first), std::move(last)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::ubjson).sax_parse(input_format_t::ubjson, &sdp, strict)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::ubjson).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } @@ -5733,7 +5733,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec auto ia = i.get(); detail::json_sax_dom_parser sdp(result, allow_exceptions); // NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg) - if (!binary_reader(std::move(ia), input_format_t::ubjson).sax_parse(input_format_t::ubjson, &sdp, strict)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::ubjson).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } @@ -5751,7 +5751,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::forward(i)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::bjdata).sax_parse(input_format_t::bjdata, &sdp, strict)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::bjdata).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } @@ -5770,7 +5770,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::move(first), std::move(last)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::bjdata).sax_parse(input_format_t::bjdata, &sdp, strict)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::bjdata).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } @@ -5788,7 +5788,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::forward(i)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::bon8).sax_parse(input_format_t::bon8, &sdp, strict)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::bon8).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } @@ -5807,7 +5807,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::move(first), std::move(last)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::bon8).sax_parse(input_format_t::bon8, &sdp, strict)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::bon8).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } @@ -5825,7 +5825,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::forward(i)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::bson).sax_parse(input_format_t::bson, &sdp, strict)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::bson).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } @@ -5844,7 +5844,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::move(first), std::move(last)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::bson).sax_parse(input_format_t::bson, &sdp, strict)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::bson).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } @@ -5871,7 +5871,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec auto ia = i.get(); detail::json_sax_dom_parser sdp(result, allow_exceptions); // NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg) - if (!binary_reader(std::move(ia), input_format_t::bson).sax_parse(input_format_t::bson, &sdp, strict)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::bson).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 3437b5408..c6c4d81b5 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -3669,7 +3669,6 @@ NLOHMANN_JSON_NAMESPACE_END -#include // array #include // size_t #include // conditional, enable_if, false_type, integral_constant, is_constructible, is_integral, is_same, remove_cv, remove_reference, true_type #include // index_sequence, make_index_sequence, index_sequence_for @@ -3822,12 +3821,6 @@ struct static_const constexpr T static_const::value; #endif -template -constexpr std::array make_array(Args&& ... args) -{ - return std::array {{static_cast(std::forward(args))...}}; -} - } // namespace detail NLOHMANN_JSON_NAMESPACE_END @@ -7549,7 +7542,6 @@ NLOHMANN_JSON_NAMESPACE_END -#include // generate_n #include // array #include // ldexp #include // size_t @@ -13681,16 +13673,16 @@ class binary_reader ~binary_reader() = default; /*! - @param[in] format the binary format to parse + @brief parse in the format the constructor was given + @param[in] sax_ a SAX event processor @param[in] strict whether to expect the input to be consumed completed @param[in] tag_handler how to treat CBOR tags @return whether parsing was successful */ - JSON_HEDLEY_NON_NULL(3) - bool sax_parse(const input_format_t format, - json_sax_t* sax_, + JSON_HEDLEY_NON_NULL(2) + bool sax_parse(json_sax_t* sax_, const bool strict = true, const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error) { @@ -13699,14 +13691,14 @@ class binary_reader bon8_pushback_size = 0; bool result = false; - switch (format) + switch (input_format) { case input_format_t::bson: result = parse_bson_internal(); break; case input_format_t::cbor: - result = parse_cbor_internal(true, tag_handler); + result = parse_cbor_internal(tag_handler); break; case input_format_t::msgpack: @@ -13829,6 +13821,22 @@ class binary_reader return enter_container(/*is_object*/true, len, type_marker); } + /*! + @brief close the innermost open array or object + + Pops the container opened by the matching @ref enter_container call and + emits the SAX end event. Every format-specific driver otherwise repeated + the same pop-then-dispatch sequence at its own close site. + + @return whether the SAX parser accepted the end event + */ + bool leave_container() + { + const bool is_object = container_stack.back().is_object; + container_stack.pop_back(); + return is_object ? sax->end_object() : sax->end_array(); + } + ////////// // BSON // ////////// @@ -13913,8 +13921,8 @@ class binary_reader if (element_type == 0) // end of the innermost document { // a copy, not a reference: it must stay valid across the - // pop_back() below, which destroys the container_stack - // element it would otherwise alias + // pop_back() inside leave_container() below, which destroys + // the container_stack element it would otherwise alias const container_frame top = container_stack.back(); if (JSON_HEDLEY_UNLIKELY(!check_bson_document_size(top.start_position, top.declared_size))) @@ -13922,8 +13930,7 @@ class binary_reader return false; } - container_stack.pop_back(); - if (JSON_HEDLEY_UNLIKELY(top.is_object ? !sax->end_object() : !sax->end_array())) + if (JSON_HEDLEY_UNLIKELY(!leave_container())) { return false; } @@ -14418,29 +14425,13 @@ class binary_reader return enter_array(conditional_static_cast(static_cast(current) & 0x1Fu)); case 0x98: // array (one-byte uint8_t for n follows) - { - std::uint8_t len{}; - return get_number(input_format_t::cbor, len) && enter_array(static_cast(len)); - } - case 0x99: // array (two-byte uint16_t for n follow) - { - std::uint16_t len{}; - return get_number(input_format_t::cbor, len) && enter_array(static_cast(len)); - } - case 0x9A: // array (four-byte uint32_t for n follow) - { - std::uint32_t len{}; - std::size_t size{}; - return get_number(input_format_t::cbor, len) && get_cbor_container_size(len, size, "array") && enter_array(size); - } - case 0x9B: // array (eight-byte uint64_t for n follow) { std::uint64_t len{}; std::size_t size{}; - return get_number(input_format_t::cbor, len) && get_cbor_container_size(len, size, "array") && enter_array(size); + return get_cbor_argument(len) && get_cbor_container_size(len, size, "array") && enter_array(size); } case 0x9F: // array (indefinite length) @@ -14474,35 +14465,19 @@ class binary_reader return enter_object(conditional_static_cast(static_cast(current) & 0x1Fu)); case 0xB8: // map (one-byte uint8_t for n follows) - { - std::uint8_t len{}; - return get_number(input_format_t::cbor, len) && enter_object(static_cast(len)); - } - case 0xB9: // map (two-byte uint16_t for n follow) - { - std::uint16_t len{}; - return get_number(input_format_t::cbor, len) && enter_object(static_cast(len)); - } - case 0xBA: // map (four-byte uint32_t for n follow) - { - std::uint32_t len{}; - std::size_t size{}; - return get_number(input_format_t::cbor, len) && get_cbor_container_size(len, size, "map") && enter_object(size); - } - case 0xBB: // map (eight-byte uint64_t for n follow) { std::uint64_t len{}; std::size_t size{}; - return get_number(input_format_t::cbor, len) && get_cbor_container_size(len, size, "map") && enter_object(size); + return get_cbor_argument(len) && get_cbor_container_size(len, size, "map") && enter_object(size); } case 0xBF: // map (indefinite length) return enter_object(detail::unknown_size()); - case 0xC0: // tagged item + case 0xC0: // tagged item (tag value 0-23, in the head itself) case 0xC1: case 0xC2: case 0xC3: @@ -14526,6 +14501,22 @@ class binary_reader case 0xD5: case 0xD6: case 0xD7: + { + if (tag_handler == cbor_tag_handler_t::error) + { + auto last_token = get_token_string(); + return sax->parse_error(chars_read, last_token, parse_error::create(112, chars_read, + exception_message(input_format_t::cbor, concat("invalid byte: 0x", last_token), "value"), nullptr)); + } + + // ignore and store: the tag value is already in the head, so + // there is nothing left to read here; the tagged value that + // follows is read by the loop in parse_cbor_internal() rather + // than by recursing here + tag_pending = true; + return true; + } + case 0xD8: // tagged item (1 byte follows) case 0xD9: // tagged item (2 bytes follow) case 0xDA: // tagged item (4 bytes follow) @@ -14542,47 +14533,11 @@ class binary_reader case cbor_tag_handler_t::ignore: { - // ignore binary subtype - switch (current) + // ignore the tag's binary subtype argument + std::uint64_t subtype_to_ignore{}; + if (!get_cbor_argument(subtype_to_ignore)) { - case 0xD8: - { - std::uint8_t subtype_to_ignore{}; - if (!get_number(input_format_t::cbor, subtype_to_ignore)) - { - return false; - } - break; - } - case 0xD9: - { - std::uint16_t subtype_to_ignore{}; - if (!get_number(input_format_t::cbor, subtype_to_ignore)) - { - return false; - } - break; - } - case 0xDA: - { - std::uint32_t subtype_to_ignore{}; - if (!get_number(input_format_t::cbor, subtype_to_ignore)) - { - return false; - } - break; - } - case 0xDB: - { - std::uint64_t subtype_to_ignore{}; - if (!get_number(input_format_t::cbor, subtype_to_ignore)) - { - return false; - } - break; - } - default: - break; + return false; } // the tagged value follows; it is read by the loop in // parse_cbor_internal() rather than by recursing here @@ -14592,57 +14547,15 @@ class binary_reader case cbor_tag_handler_t::store: { - binary_t b; // use binary subtype and store in a binary container - switch (current) + std::uint64_t subtype{}; + if (!get_cbor_argument(subtype)) { - case 0xD8: - { - std::uint8_t subtype{}; - if (!get_number(input_format_t::cbor, subtype)) - { - return false; - } - b.set_subtype(detail::conditional_static_cast(subtype)); - break; - } - case 0xD9: - { - std::uint16_t subtype{}; - if (!get_number(input_format_t::cbor, subtype)) - { - return false; - } - b.set_subtype(detail::conditional_static_cast(subtype)); - break; - } - case 0xDA: - { - std::uint32_t subtype{}; - if (!get_number(input_format_t::cbor, subtype)) - { - return false; - } - b.set_subtype(detail::conditional_static_cast(subtype)); - break; - } - case 0xDB: - { - std::uint64_t subtype{}; - if (!get_number(input_format_t::cbor, subtype)) - { - return false; - } - b.set_subtype(detail::conditional_static_cast(subtype)); - break; - } - default: - { - // as above, the tagged value is read by the caller - tag_pending = true; - return true; - } + return false; } + binary_t b; + b.set_subtype(detail::conditional_static_cast(subtype)); + get(); // a byte string (the heads accepted by get_cbor_binary) keeps the tag as subtype if ((current >= 0x40 && current <= 0x5B) || current == 0x5F) @@ -14673,52 +14586,7 @@ class binary_reader return sax->null(); case 0xF9: // Half-Precision Float (two-byte IEEE 754) - { - const auto byte1_raw = get(); - if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format_t::cbor, "number"))) - { - return false; - } - const auto byte2_raw = get(); - if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format_t::cbor, "number"))) - { - return false; - } - - const auto byte1 = static_cast(byte1_raw); - const auto byte2 = static_cast(byte2_raw); - - // Code from RFC 8949, Appendix D, Figure 3: - // As half-precision floating-point numbers were only added - // to IEEE 754 in 2008, today's programming platforms often - // still only have limited support for them. It is very - // easy to include at least decoding support for them even - // without such support. An example of a small decoder for - // half-precision floating-point numbers in the C language - // is shown in Fig. 3. - const auto half = static_cast((byte1 << 8u) + byte2); - const double val = [&half] - { - const int exp = (half >> 10u) & 0x1Fu; - const unsigned int mant = half & 0x3FFu; - JSON_ASSERT(exp <= 31); - JSON_ASSERT(mant <= 1023); - switch (exp) - { - case 0: - return std::ldexp(mant, -24); - case 31: - return (mant == 0) - ? std::numeric_limits::infinity() - : std::numeric_limits::quiet_NaN(); - default: - return std::ldexp(mant + 1024, exp - 25); - } - }(); - return sax->number_float((half & 0x8000u) != 0 - ? static_cast(-val) - : static_cast(val), ""); - } + return get_half_float(input_format_t::cbor, false); case 0xFA: // Single-Precision Float (four-byte IEEE 754) { @@ -15097,6 +14965,73 @@ class binary_reader } } + /*! + @brief read a CBOR argument (additional information 24-27) of the width + @ref current announces + + The lower 5 bits of @a current (0x18-0x1B) select a 1/2/4/8-byte + big-endian unsigned integer that follows the head byte; this is shared by + every major type that uses this encoding (unsigned/negative integers, + strings, arrays, maps, tags). Reading always goes through @ref get_number, + so EOF is reported the same way as before this helper existed. + + @param[out] value the decoded argument + @return whether reading succeeded + */ + bool get_cbor_argument(std::uint64_t& value) + { + switch (current & 0x1F) + { + case 0x18: // 1 byte + { + std::uint8_t n{}; + if (JSON_HEDLEY_UNLIKELY(!get_number(input_format_t::cbor, n))) + { + return false; + } + value = n; + return true; + } + + case 0x19: // 2 bytes + { + std::uint16_t n{}; + if (JSON_HEDLEY_UNLIKELY(!get_number(input_format_t::cbor, n))) + { + return false; + } + value = n; + return true; + } + + case 0x1A: // 4 bytes + { + std::uint32_t n{}; + if (JSON_HEDLEY_UNLIKELY(!get_number(input_format_t::cbor, n))) + { + return false; + } + value = n; + return true; + } + + case 0x1B: // 8 bytes + { + std::uint64_t n{}; + if (JSON_HEDLEY_UNLIKELY(!get_number(input_format_t::cbor, n))) + { + return false; + } + value = n; + return true; + } + + default: // LCOV_EXCL_LINE + JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE + return false; // LCOV_EXCL_LINE + } + } + /*! @brief narrow a definite CBOR array/map length to std::size_t @@ -15129,19 +15064,15 @@ class binary_reader enclosing container after each element, so that the nesting depth of the input costs heap rather than native stack (see #5104). - @param[in] get_char whether a new character should be retrieved from the - input (true) or whether the last read character - @a current should be considered instead @param[in] tag_handler how CBOR tags should be treated @return whether reading the value succeeded */ - bool parse_cbor_internal(const bool get_char, - const cbor_tag_handler_t tag_handler) + bool parse_cbor_internal(const cbor_tag_handler_t tag_handler) { // whether the next value starts at a fresh byte or at the one already // read into `current` - bool fetch = get_char; + bool fetch = true; // the key currently being read; hoisted out of the loop so that its // capacity is reused across elements and across nesting levels @@ -15184,8 +15115,7 @@ class binary_reader if (at_end) { - container_stack.pop_back(); - if (JSON_HEDLEY_UNLIKELY(top.is_object ? !sax->end_object() : !sax->end_array())) + if (JSON_HEDLEY_UNLIKELY(!leave_container())) { return false; } @@ -15234,9 +15164,6 @@ class binary_reader // MsgPack // ///////////// - /*! - @return whether a valid MessagePack value was passed to the SAX parser - */ /*! @brief read one MessagePack value @@ -15935,8 +15862,7 @@ class binary_reader if (container_stack.back().remaining == 0) { - container_stack.pop_back(); - if (JSON_HEDLEY_UNLIKELY(is_object ? !sax->end_object() : !sax->end_array())) + if (JSON_HEDLEY_UNLIKELY(!leave_container())) { return false; } @@ -15981,20 +15907,16 @@ class binary_reader //////////// /*! - @param[in] get_char whether a new character should be retrieved from the - input (true, default) or whether the last read - character should be considered instead - @return whether a valid UBJSON value was passed to the SAX parser */ - bool parse_ubjson_internal(const bool get_char = true) + bool parse_ubjson_internal() { // the key currently being read; hoisted out of the loop so that its // capacity is reused across elements and across nesting levels string_t key; // the type marker of the value to read next - char_int_type prefix = get_char ? get_ignore_noop() : current; + char_int_type prefix = get_ignore_noop(); while (true) { @@ -16069,8 +15991,7 @@ class binary_reader break; } - container_stack.pop_back(); - if (JSON_HEDLEY_UNLIKELY(top.is_object ? !sax->end_object() : !sax->end_array())) + if (JSON_HEDLEY_UNLIKELY(!leave_container())) { return false; } @@ -16278,6 +16199,42 @@ class binary_reader return true; } + /*! + @brief read a UBJSON/BJData optimized-container count of a signed marker + type ('i', 'I', 'l', 'L') and narrow it to std::size_t + + Every signed count marker rejects a negative value the same way (error + 113); the value_in_range_of check additionally needed for 'L' is only + ever live when @a SignedType is std::int64_t on a target where + std::size_t is narrower (e.g. 32-bit), since 'i'/'I'/'l' can never exceed + std::size_t there. + + @tparam SignedType std::int8_t, std::int16_t, std::int32_t or std::int64_t + @param[out] result the count narrowed to std::size_t + @return whether reading and validating succeeded + */ + template + bool get_ubjson_signed_count(std::size_t& result) + { + SignedType number{}; + if (JSON_HEDLEY_UNLIKELY(!get_number(input_format, number))) + { + return false; + } + if (JSON_HEDLEY_UNLIKELY(number < 0)) + { + return sax->parse_error(chars_read, get_token_string(), parse_error::create(113, chars_read, + exception_message(input_format, "count in an optimized container must be positive", "size"), nullptr)); + } + if (JSON_HEDLEY_UNLIKELY(!value_in_range_of(number))) + { + return sax->parse_error(chars_read, get_token_string(), out_of_range::create(408, + exception_message(input_format, "integer value overflow", "size"), nullptr)); + } + result = static_cast(number); // NOLINT(bugprone-signed-char-misuse,cert-str34-c): number is not a char + return true; + } + /*! @param[out] result determined size @param[in,out] is_ndarray for input, `true` means already inside an ndarray vector @@ -16310,73 +16267,16 @@ class binary_reader } case 'i': - { - std::int8_t number{}; - if (JSON_HEDLEY_UNLIKELY(!get_number(input_format, number))) - { - return false; - } - if (number < 0) - { - return sax->parse_error(chars_read, get_token_string(), parse_error::create(113, chars_read, - exception_message(input_format, "count in an optimized container must be positive", "size"), nullptr)); - } - result = static_cast(number); // NOLINT(bugprone-signed-char-misuse,cert-str34-c): number is not a char - return true; - } + return get_ubjson_signed_count(result); case 'I': - { - std::int16_t number{}; - if (JSON_HEDLEY_UNLIKELY(!get_number(input_format, number))) - { - return false; - } - if (number < 0) - { - return sax->parse_error(chars_read, get_token_string(), parse_error::create(113, chars_read, - exception_message(input_format, "count in an optimized container must be positive", "size"), nullptr)); - } - result = static_cast(number); - return true; - } + return get_ubjson_signed_count(result); case 'l': - { - std::int32_t number{}; - if (JSON_HEDLEY_UNLIKELY(!get_number(input_format, number))) - { - return false; - } - if (number < 0) - { - return sax->parse_error(chars_read, get_token_string(), parse_error::create(113, chars_read, - exception_message(input_format, "count in an optimized container must be positive", "size"), nullptr)); - } - result = static_cast(number); - return true; - } + return get_ubjson_signed_count(result); case 'L': - { - std::int64_t number{}; - if (JSON_HEDLEY_UNLIKELY(!get_number(input_format, number))) - { - return false; - } - if (number < 0) - { - return sax->parse_error(chars_read, get_token_string(), parse_error::create(113, chars_read, - exception_message(input_format, "count in an optimized container must be positive", "size"), nullptr)); - } - if (!value_in_range_of(number)) - { - return sax->parse_error(chars_read, get_token_string(), out_of_range::create(408, - exception_message(input_format, "integer value overflow", "size"), nullptr)); - } - result = static_cast(number); - return true; - } + return get_ubjson_signed_count(result); case 'u': { @@ -16467,16 +16367,23 @@ class binary_reader result = 1; for (auto i : dim) { - // Pre-multiplication overflow check: if i > 0 and result > SIZE_MAX/i, then result*i would overflow. - // This check must happen before multiplication since overflow detection after the fact is unreliable - // as modular arithmetic can produce any value, not just 0 or SIZE_MAX. - if (JSON_HEDLEY_UNLIKELY(i > 0 && result > (std::numeric_limits::max)() / i)) + // Pre-multiplication overflow check: since the loop above + // already rejected any zero dimension, i is always > 0 + // here, so result > SIZE_MAX/i means result*i would + // overflow. This check must happen before multiplication + // since overflow detection after the fact is unreliable, + // as modular arithmetic can produce any value, not just 0 + // or SIZE_MAX. + if (JSON_HEDLEY_UNLIKELY(result > (std::numeric_limits::max)() / i)) { return sax->parse_error(chars_read, get_token_string(), out_of_range::create(408, exception_message(input_format, "excessive ndarray size caused overflow", "size"), nullptr)); } result *= i; - // Additional post-multiplication check to catch any edge cases the pre-check might miss - if (result == 0 || result == npos) + // the pre-check above already rules out result becoming 0 + // by overflow; the only value it cannot rule out is an + // exact match with npos, the sentinel reserved for an + // unknown-size container (see get_ubjson_size_type()) + if (result == npos) { return sax->parse_error(chars_read, get_token_string(), out_of_range::create(408, exception_message(input_format, "excessive ndarray size caused overflow", "size"), nullptr)); } @@ -16537,7 +16444,7 @@ class binary_reader { result.second = get(); // must not ignore 'N', because 'N' maybe the type if (input_format == input_format_t::bjdata - && JSON_HEDLEY_UNLIKELY(std::binary_search(bjd_optimized_type_markers.begin(), bjd_optimized_type_markers.end(), result.second))) + && JSON_HEDLEY_UNLIKELY(is_bjd_excluded_optimized_type(result.second))) { auto last_token = get_token_string(); return sax->parse_error(chars_read, last_token, parse_error::create(112, chars_read, @@ -16681,50 +16588,7 @@ class binary_reader { break; } - const auto byte1_raw = get(); - if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format, "number"))) - { - return false; - } - const auto byte2_raw = get(); - if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format, "number"))) - { - return false; - } - - const auto byte1 = static_cast(byte1_raw); - const auto byte2 = static_cast(byte2_raw); - - // Code from RFC 8949, Appendix D, Figure 3: - // As half-precision floating-point numbers were only added - // to IEEE 754 in 2008, today's programming platforms often - // still only have limited support for them. It is very - // easy to include at least decoding support for them even - // without such support. An example of a small decoder for - // half-precision floating-point numbers in the C language - // is shown in Fig. 3. - const auto half = static_cast((byte2 << 8u) + byte1); - const double val = [&half] - { - const int exp = (half >> 10u) & 0x1Fu; - const unsigned int mant = half & 0x3FFu; - JSON_ASSERT(exp <= 31); - JSON_ASSERT(mant <= 1023); - switch (exp) - { - case 0: - return std::ldexp(mant, -24); - case 31: - return (mant == 0) - ? std::numeric_limits::infinity() - : std::numeric_limits::quiet_NaN(); - default: - return std::ldexp(mant + 1024, exp - 25); - } - }(); - return sax->number_float((half & 0x8000u) != 0 - ? static_cast(-val) - : static_cast(val), ""); + return get_half_float(input_format, true); } case 'd': @@ -16797,19 +16661,16 @@ class binary_reader if (input_format == input_format_t::bjdata && size_and_type.first != npos && (size_and_type.second & (1 << 8)) != 0) { size_and_type.second &= ~(static_cast(1) << 8); // use bit 8 to indicate ndarray, here we remove the bit to restore the type marker - auto it = std::lower_bound(bjd_types_map.begin(), bjd_types_map.end(), size_and_type.second, [](const bjd_type & p, char_int_type t) - { - return p.first < t; - }); + const char* type_name = bjd_type_name(size_and_type.second); string_t key = "_ArrayType_"; - if (JSON_HEDLEY_UNLIKELY(it == bjd_types_map.end() || it->first != size_and_type.second)) + if (JSON_HEDLEY_UNLIKELY(type_name == nullptr)) { auto last_token = get_token_string(); return sax->parse_error(chars_read, last_token, parse_error::create(112, chars_read, exception_message(input_format, "invalid byte: 0x" + last_token, "type"), nullptr)); } - string_t type = it->second; // sax->string() takes a reference + string_t type = type_name; // sax->string() takes a reference if (JSON_HEDLEY_UNLIKELY(!sax->key(key) || !sax->string(type))) { return false; @@ -17101,8 +16962,7 @@ class binary_reader if (at_end) { - container_stack.pop_back(); - if (JSON_HEDLEY_UNLIKELY(top.is_object ? !sax->end_object() : !sax->end_array())) + if (JSON_HEDLEY_UNLIKELY(!leave_container())) { return false; } @@ -17641,6 +17501,68 @@ class binary_reader return true; } + /*! + @brief read and decode an IEEE 754 half-precision (16-bit) float + + Used by CBOR (big endian) and BJData (little endian); the two formats + only differ in the byte order of the two bytes that make up the half. + + @param[in] format the current format (for diagnostics) + @param[in] little_endian whether the two bytes are little endian (BJData) + or big endian (CBOR) + + @return whether reading and decoding succeeded + */ + bool get_half_float(const input_format_t format, const bool little_endian) + { + const auto byte1_raw = get(); + if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(format, "number"))) + { + return false; + } + const auto byte2_raw = get(); + if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(format, "number"))) + { + return false; + } + + const auto byte1 = static_cast(byte1_raw); + const auto byte2 = static_cast(byte2_raw); + + // Code from RFC 8949, Appendix D, Figure 3: + // As half-precision floating-point numbers were only added + // to IEEE 754 in 2008, today's programming platforms often + // still only have limited support for them. It is very + // easy to include at least decoding support for them even + // without such support. An example of a small decoder for + // half-precision floating-point numbers in the C language + // is shown in Fig. 3. + const auto half = little_endian + ? static_cast((byte2 << 8u) + byte1) + : static_cast((byte1 << 8u) + byte2); + const double val = [&half] + { + const int exp = (half >> 10u) & 0x1Fu; + const unsigned int mant = half & 0x3FFu; + JSON_ASSERT(exp <= 31); + JSON_ASSERT(mant <= 1023); + switch (exp) + { + case 0: + return std::ldexp(mant, -24); + case 31: + return (mant == 0) + ? std::numeric_limits::infinity() + : std::numeric_limits::quiet_NaN(); + default: + return std::ldexp(mant + 1024, exp - 25); + } + }(); + return sax->number_float((half & 0x8000u) != 0 + ? static_cast(-val) + : static_cast(val), ""); + } + /*! @brief create a string by reading characters from the input @@ -17866,38 +17788,61 @@ class binary_reader /// BON8: number of bytes in @ref bon8_pushback std::size_t bon8_pushback_size = 0; - // excluded markers in bjdata optimized type -#define JSON_BINARY_READER_MAKE_BJD_OPTIMIZED_TYPE_MARKERS_ \ - make_array('F', 'H', 'N', 'S', 'T', 'Z', '[', '{') - -#define JSON_BINARY_READER_MAKE_BJD_TYPES_MAP_ \ - make_array( \ - bjd_type{'B', "byte"}, \ - bjd_type{'C', "char"}, \ - bjd_type{'D', "double"}, \ - bjd_type{'I', "int16"}, \ - bjd_type{'L', "int64"}, \ - bjd_type{'M', "uint64"}, \ - bjd_type{'U', "uint8"}, \ - bjd_type{'d', "single"}, \ - bjd_type{'i', "int8"}, \ - bjd_type{'l', "int32"}, \ - bjd_type{'m', "uint32"}, \ - bjd_type{'u', "uint16"}) - JSON_PRIVATE_UNLESS_TESTED: - // lookup tables - // NOLINTNEXTLINE(cppcoreguidelines-non-private-member-variables-in-classes) - const decltype(JSON_BINARY_READER_MAKE_BJD_OPTIMIZED_TYPE_MARKERS_) bjd_optimized_type_markers = - JSON_BINARY_READER_MAKE_BJD_OPTIMIZED_TYPE_MARKERS_; + /*! + @brief whether @a marker is excluded from BJData's optimized ND-array types + @return whether @a marker is one of 'F', 'H', 'N', 'S', 'T', 'Z', '[', '{' - using bjd_type = std::pair; - // NOLINTNEXTLINE(cppcoreguidelines-non-private-member-variables-in-classes) - const decltype(JSON_BINARY_READER_MAKE_BJD_TYPES_MAP_) bjd_types_map = - JSON_BINARY_READER_MAKE_BJD_TYPES_MAP_; + Mirrors binary_writer's @ref binary_writer::is_bjdata_excluded_type_marker + "is_bjdata_excluded_type_marker()`, which encodes the same list the other + way; keep the two in sync. + */ + static constexpr bool is_bjd_excluded_optimized_type(const char_int_type marker) noexcept + { + return marker == '[' || marker == '{' || marker == 'S' || marker == 'H' + || marker == 'T' || marker == 'F' || marker == 'N' || marker == 'Z'; + } -#undef JSON_BINARY_READER_MAKE_BJD_OPTIMIZED_TYPE_MARKERS_ -#undef JSON_BINARY_READER_MAKE_BJD_TYPES_MAP_ + /*! + @brief look up the ND-array element type name for a BJData dtype marker + @return the type name ("uint8", "int8", ...), or nullptr if @a marker does + not name a known dtype + + A C++11 `constexpr` function cannot contain a `switch`, so this is a + plain (non-constexpr) switch instead. + */ + static const char* bjd_type_name(const char_int_type marker) + { + switch (marker) + { + case 'B': + return "byte"; + case 'C': + return "char"; + case 'D': + return "double"; + case 'I': + return "int16"; + case 'L': + return "int64"; + case 'M': + return "uint64"; + case 'U': + return "uint8"; + case 'd': + return "single"; + case 'i': + return "int8"; + case 'l': + return "int32"; + case 'm': + return "uint32"; + case 'u': + return "uint16"; + default: + return nullptr; + } + } }; #ifndef JSON_HAS_CPP_17 @@ -20914,7 +20859,6 @@ NLOHMANN_JSON_NAMESPACE_END #include // reverse #include // array -#include // map #include // isnan, isinf #include // uint8_t, uint16_t, uint32_t, uint64_t #include // memcpy @@ -21244,7 +21188,7 @@ class binary_writer /*! @param[in] j JSON value to serialize - @pre j.type() == value_t::object + @throw type_error.317 if @a j is not an object */ void write_bson(const BasicJsonType& j) { @@ -21333,7 +21277,7 @@ class binary_writer } else { - write_compact_float(j.m_data.m_value.number_float, detail::input_format_t::cbor); + write_compact_float(j.m_data.m_value.number_float, to_char_type(0xFA), to_char_type(0xFB)); } break; } @@ -21367,6 +21311,13 @@ class binary_writer { if (j.m_data.m_value.binary->has_subtype()) { + // The subtype is always written as a tag with a 0xD8..0xDB + // head, never in the one-byte form 0xC0..0xD7 that CBOR + // allows for tags 0..23 (so this is not write_cbor_head). + // binary_reader with cbor_tag_handler_t::store only turns + // 0xD8..0xDB into a subtype and ignores the one-byte tags, + // so the shorter form would lose subtypes 0..23 on a round + // trip. if (j.m_data.m_value.binary->subtype() <= (std::numeric_limits::max)()) { write_number(static_cast(0xd8)); @@ -21438,6 +21389,43 @@ class binary_writer return static_cast(length); } + /*! + @brief write a non-negative integer using the MessagePack fixint/uint ladder + @param[in] n the value to write, already known to be non-negative + */ + void write_msgpack_unsigned(const std::uint64_t n) + { + if (n < 128) + { + // positive fixnum + write_number(static_cast(n)); + } + else if (n <= (std::numeric_limits::max)()) + { + // uint 8 + oa.write_character(to_char_type(0xCC)); + write_number(static_cast(n)); + } + else if (n <= (std::numeric_limits::max)()) + { + // uint 16 + oa.write_character(to_char_type(0xCD)); + write_number(static_cast(n)); + } + else if (n <= (std::numeric_limits::max)()) + { + // uint 32 + oa.write_character(to_char_type(0xCE)); + write_number(static_cast(n)); + } + else + { + // uint 64 + oa.write_character(to_char_type(0xCF)); + write_number(n); + } + } + /*! @param[in] j JSON value to serialize */ @@ -21464,38 +21452,8 @@ class binary_writer if (j.m_data.m_value.number_integer >= 0) { // MessagePack does not differentiate between positive - // signed integers and unsigned integers. Therefore, we used - // the code from the value_t::number_unsigned case here. - const auto value_as_unsigned = static_cast(j.m_data.m_value.number_integer); - if (value_as_unsigned < 128) - { - // positive fixnum - write_number(static_cast(j.m_data.m_value.number_integer)); - } - else if (value_as_unsigned <= (std::numeric_limits::max)()) - { - // uint 8 - oa.write_character(to_char_type(0xCC)); - write_number(static_cast(j.m_data.m_value.number_integer)); - } - else if (value_as_unsigned <= (std::numeric_limits::max)()) - { - // uint 16 - oa.write_character(to_char_type(0xCD)); - write_number(static_cast(j.m_data.m_value.number_integer)); - } - else if (value_as_unsigned <= (std::numeric_limits::max)()) - { - // uint 32 - oa.write_character(to_char_type(0xCE)); - write_number(static_cast(j.m_data.m_value.number_integer)); - } - else - { - // uint 64 - oa.write_character(to_char_type(0xCF)); - write_number(static_cast(j.m_data.m_value.number_integer)); - } + // signed integers and unsigned integers. + write_msgpack_unsigned(static_cast(j.m_data.m_value.number_integer)); } else { @@ -21537,41 +21495,13 @@ class binary_writer case value_t::number_unsigned: { - if (j.m_data.m_value.number_unsigned < 128) - { - // positive fixnum - write_number(static_cast(j.m_data.m_value.number_unsigned)); - } - else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) - { - // uint 8 - oa.write_character(to_char_type(0xCC)); - write_number(static_cast(j.m_data.m_value.number_unsigned)); - } - else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) - { - // uint 16 - oa.write_character(to_char_type(0xCD)); - write_number(static_cast(j.m_data.m_value.number_unsigned)); - } - else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) - { - // uint 32 - oa.write_character(to_char_type(0xCE)); - write_number(static_cast(j.m_data.m_value.number_unsigned)); - } - else - { - // uint 64 - oa.write_character(to_char_type(0xCF)); - write_number(static_cast(j.m_data.m_value.number_unsigned)); - } + write_msgpack_unsigned(static_cast(j.m_data.m_value.number_unsigned)); break; } case value_t::number_float: { - write_compact_float(j.m_data.m_value.number_float, detail::input_format_t::msgpack); + write_compact_float(j.m_data.m_value.number_float, to_char_type(0xCA), to_char_type(0xCB)); break; } @@ -22512,6 +22442,14 @@ class binary_writer oa.write_character(to_char_type(0x00)); if (parents.empty()) { + // calc_bson_sizes() and write_bson_document() are two + // hand-synchronized passes over the same structure, linked + // only by nested_sizes' visiting order; this checks that the + // write pass consumed exactly the sizes the size pass + // produced, so a future change that desyncs them (skips or + // rejects an entry in only one pass) is caught immediately + // instead of silently writing wrong length prefixes. + JSON_ASSERT(next_size == nested_sizes.size()); return; } current = std::move(parents.back()); @@ -22563,52 +22501,6 @@ class binary_writer } } - static constexpr CharType get_cbor_float_prefix(float /*unused*/) - { - return to_char_type(0xFA); // Single-Precision Float - } - - static constexpr CharType get_cbor_float_prefix(double /*unused*/) - { - return to_char_type(0xFB); // Double-Precision Float - } - - ///////////// - // MsgPack // - ///////////// - - static constexpr CharType get_msgpack_float_prefix(float /*unused*/) - { - return to_char_type(0xCA); // float 32 - } - - static constexpr CharType get_msgpack_float_prefix(double /*unused*/) - { - return to_char_type(0xCB); // float 64 - } - - /// @return the BON8 type marker for binary32 (float) or binary64 (double) - template - static constexpr CharType get_bon8_float_prefix() - { - return to_char_type(std::is_same::value ? 0x8E : 0x8F); - } - - /// @return the type marker for a FloatType value in @a format (CBOR, MessagePack, or BON8) - template - static CharType get_compact_float_prefix(const detail::input_format_t format) - { - if (format == detail::input_format_t::cbor) - { - return get_cbor_float_prefix(FloatType{}); - } - if (format == detail::input_format_t::bon8) - { - return get_bon8_float_prefix(); - } - return get_msgpack_float_prefix(FloatType{}); - } - //////////// // UBJSON // //////////// @@ -22622,207 +22514,127 @@ class binary_writer { if (add_prefix) { - oa.write_character(get_ubjson_float_prefix(n)); + oa.write_character(get_ubjson_float_prefix()); } write_number(n, use_bjdata); } - // UBJSON: write number (unsigned integer) + // UBJSON: write number (integer) template::value, int>::type = 0> + std::is_integral::value, int>::type = 0> void write_number_with_ubjson_prefix(const NumberType n, const bool add_prefix, const bool use_bjdata) { - if (n <= static_cast((std::numeric_limits::max)())) - { - if (add_prefix) - { - oa.write_character(to_char_type('i')); // int8 - } - write_number(static_cast(n), use_bjdata); - } - else if (n <= (std::numeric_limits::max)()) - { - if (add_prefix) - { - oa.write_character(to_char_type('U')); // uint8 - } - write_number(static_cast(n), use_bjdata); - } - else if (n <= static_cast((std::numeric_limits::max)())) - { - if (add_prefix) - { - oa.write_character(to_char_type('I')); // int16 - } - write_number(static_cast(n), use_bjdata); - } - else if (use_bjdata && n <= static_cast((std::numeric_limits::max)())) - { - if (add_prefix) - { - oa.write_character(to_char_type('u')); // uint16 - bjdata only - } - write_number(static_cast(n), use_bjdata); - } - else if (n <= static_cast((std::numeric_limits::max)())) - { - if (add_prefix) - { - oa.write_character(to_char_type('l')); // int32 - } - write_number(static_cast(n), use_bjdata); - } - else if (use_bjdata && n <= static_cast((std::numeric_limits::max)())) - { - if (add_prefix) - { - oa.write_character(to_char_type('m')); // uint32 - bjdata only - } - write_number(static_cast(n), use_bjdata); - } - else if (n <= static_cast((std::numeric_limits::max)())) - { - if (add_prefix) - { - oa.write_character(to_char_type('L')); // int64 - } - write_number(static_cast(n), use_bjdata); - } - else if (use_bjdata) - { - if (add_prefix) - { - oa.write_character(to_char_type('M')); // uint64 - bjdata only - } - write_number(static_cast(n), use_bjdata); - } - else - { - if (add_prefix) - { - oa.write_character(to_char_type('H')); // high-precision number - } - - const auto number = BasicJsonType(n).dump(); - write_number_with_ubjson_prefix(number.size(), true, use_bjdata); - for (std::size_t i = 0; i < number.size(); ++i) - { - oa.write_character(to_char_type(static_cast(number[i]))); - } - } - } - - // UBJSON: write number (signed integer) - template < typename NumberType, typename std::enable_if < - std::is_signed::value&& - !std::is_floating_point::value, int >::type = 0 > - void write_number_with_ubjson_prefix(const NumberType n, - const bool add_prefix, - const bool use_bjdata) - { - if ((std::numeric_limits::min)() <= n && n <= (std::numeric_limits::max)()) - { - if (add_prefix) - { - oa.write_character(to_char_type('i')); // int8 - } - write_number(static_cast(n), use_bjdata); - } - else if (static_cast((std::numeric_limits::min)()) <= n && n <= static_cast((std::numeric_limits::max)())) - { - if (add_prefix) - { - oa.write_character(to_char_type('U')); // uint8 - } - write_number(static_cast(n), use_bjdata); - } - else if ((std::numeric_limits::min)() <= n && n <= (std::numeric_limits::max)()) - { - if (add_prefix) - { - oa.write_character(to_char_type('I')); // int16 - } - write_number(static_cast(n), use_bjdata); - } - else if (use_bjdata && (static_cast((std::numeric_limits::min)()) <= n && n <= static_cast((std::numeric_limits::max)()))) - { - if (add_prefix) - { - oa.write_character(to_char_type('u')); // uint16 - bjdata only - } - write_number(static_cast(n), use_bjdata); - } - else if ((std::numeric_limits::min)() <= n && n <= (std::numeric_limits::max)()) - { - if (add_prefix) - { - oa.write_character(to_char_type('l')); // int32 - } - write_number(static_cast(n), use_bjdata); - } - else if (use_bjdata && (static_cast((std::numeric_limits::min)()) <= n && n <= static_cast((std::numeric_limits::max)()))) - { - if (add_prefix) - { - oa.write_character(to_char_type('m')); // uint32 - bjdata only - } - write_number(static_cast(n), use_bjdata); - } - else - { - // every value of an integer type of at most 64 bits fits into an - // int64; only a wider type needs a range check - write_ubjson_int64_or_high_precision(n, add_prefix, use_bjdata, - std::integral_constant < bool, std::numeric_limits::digits <= std::numeric_limits::digits > {}); - } - } - - template - void write_ubjson_int64_or_high_precision(const NumberType n, const bool add_prefix, const bool use_bjdata, std::true_type /*fits_int64*/) - { + const CharType prefix = ubjson_integer_prefix(n, use_bjdata); if (add_prefix) { - oa.write_character(to_char_type('L')); // int64 + oa.write_character(prefix); } - write_number(static_cast(n), use_bjdata); + write_ubjson_integer_payload(prefix, n, use_bjdata); } + /*! + @brief determine the UBJSON/BJData type marker of an integer + + This is the only place that picks the marker of an integer: both + write_number_with_ubjson_prefix() and ubjson_prefix() use it. An optimized + container announces the marker of its first value after `$` and then + writes every value without a marker, so the two must never disagree. + + @param[in] n the integer + @param[in] use_bjdata whether the BJData-only markers `u`, `m`, and `M` + may be used + + @return the first marker of `i`, `U`, `I`, `u` (BJData), `l`, `m` (BJData), + `L`, `M` (BJData, unsigned types only), and `H` (high-precision + number) whose range contains @a n + */ template - void write_ubjson_int64_or_high_precision(const NumberType n, const bool add_prefix, const bool use_bjdata, std::false_type /*fits_int64*/) + static CharType ubjson_integer_prefix(const NumberType n, const bool use_bjdata) noexcept { - if ((std::numeric_limits::min)() <= n && n <= (std::numeric_limits::max)()) + if (value_in_range_of(n)) { - write_ubjson_int64_or_high_precision(n, add_prefix, use_bjdata, std::true_type {}); - return; + return 'i'; } - - if (add_prefix) + if (value_in_range_of(n)) { - oa.write_character(to_char_type('H')); // high-precision number + return 'U'; } - - const auto number = BasicJsonType(n).dump(); - write_number_with_ubjson_prefix(number.size(), true, use_bjdata); - for (std::size_t i = 0; i < number.size(); ++i) + if (value_in_range_of(n)) { - oa.write_character(to_char_type(static_cast(number[i]))); + return 'I'; } + if (use_bjdata && value_in_range_of(n)) + { + return 'u'; + } + if (value_in_range_of(n)) + { + return 'l'; + } + if (use_bjdata && value_in_range_of(n)) + { + return 'm'; + } + if (value_in_range_of(n)) + { + return 'L'; + } + if (use_bjdata && std::is_unsigned::value) + { + return 'M'; + } + // anything else is treated as a high-precision number + return 'H'; } + /*! + @brief write the value of an integer for the marker chosen by + ubjson_integer_prefix() + */ template - static constexpr CharType ubjson_int64_or_high_precision_prefix(const NumberType /*n*/, std::true_type /*fits_int64*/) noexcept + void write_ubjson_integer_payload(const CharType prefix, const NumberType n, const bool use_bjdata) { - return 'L'; - } - - template - static CharType ubjson_int64_or_high_precision_prefix(const NumberType n, std::false_type /*fits_int64*/) noexcept - { - // anything outside of the range of an int64 is treated as a - // high-precision number - return ((std::numeric_limits::min)() <= n && n <= (std::numeric_limits::max)()) ? 'L' : 'H'; + switch (prefix) + { + case 'i': + write_number(static_cast(n), use_bjdata); + break; + case 'U': + write_number(static_cast(n), use_bjdata); + break; + case 'I': + write_number(static_cast(n), use_bjdata); + break; + case 'u': + write_number(static_cast(n), use_bjdata); + break; + case 'l': + write_number(static_cast(n), use_bjdata); + break; + case 'm': + write_number(static_cast(n), use_bjdata); + break; + case 'L': + write_number(static_cast(n), use_bjdata); + break; + case 'M': + write_number(static_cast(n), use_bjdata); + break; + default: + { + // high-precision number: the decimal digits as a string + JSON_ASSERT(prefix == 'H'); + const auto number = BasicJsonType(n).dump(); + write_number_with_ubjson_prefix(number.size(), true, use_bjdata); + for (std::size_t i = 0; i < number.size(); ++i) + { + oa.write_character(to_char_type(static_cast(number[i]))); + } + break; + } + } } /*! @@ -22839,77 +22651,13 @@ class binary_writer return j.m_data.m_value.boolean ? 'T' : 'F'; case value_t::number_integer: - { - if ((std::numeric_limits::min)() <= j.m_data.m_value.number_integer && j.m_data.m_value.number_integer <= (std::numeric_limits::max)()) - { - return 'i'; - } - if ((std::numeric_limits::min)() <= j.m_data.m_value.number_integer && j.m_data.m_value.number_integer <= (std::numeric_limits::max)()) - { - return 'U'; - } - if ((std::numeric_limits::min)() <= j.m_data.m_value.number_integer && j.m_data.m_value.number_integer <= (std::numeric_limits::max)()) - { - return 'I'; - } - if (use_bjdata && ((std::numeric_limits::min)() <= j.m_data.m_value.number_integer && j.m_data.m_value.number_integer <= (std::numeric_limits::max)())) - { - return 'u'; - } - if ((std::numeric_limits::min)() <= j.m_data.m_value.number_integer && j.m_data.m_value.number_integer <= (std::numeric_limits::max)()) - { - return 'l'; - } - if (use_bjdata && ((std::numeric_limits::min)() <= j.m_data.m_value.number_integer && j.m_data.m_value.number_integer <= (std::numeric_limits::max)())) - { - return 'm'; - } - // every value of an integer type of at most 64 bits fits into - // an int64; only a wider type needs a range check - return ubjson_int64_or_high_precision_prefix(j.m_data.m_value.number_integer, - std::integral_constant < bool, std::numeric_limits::digits <= std::numeric_limits::digits > {}); - } + return ubjson_integer_prefix(j.m_data.m_value.number_integer, use_bjdata); case value_t::number_unsigned: - { - if (j.m_data.m_value.number_unsigned <= static_cast((std::numeric_limits::max)())) - { - return 'i'; - } - if (j.m_data.m_value.number_unsigned <= static_cast((std::numeric_limits::max)())) - { - return 'U'; - } - if (j.m_data.m_value.number_unsigned <= static_cast((std::numeric_limits::max)())) - { - return 'I'; - } - if (use_bjdata && j.m_data.m_value.number_unsigned <= static_cast((std::numeric_limits::max)())) - { - return 'u'; - } - if (j.m_data.m_value.number_unsigned <= static_cast((std::numeric_limits::max)())) - { - return 'l'; - } - if (use_bjdata && j.m_data.m_value.number_unsigned <= static_cast((std::numeric_limits::max)())) - { - return 'm'; - } - if (j.m_data.m_value.number_unsigned <= static_cast((std::numeric_limits::max)())) - { - return 'L'; - } - if (use_bjdata) - { - return 'M'; - } - // anything else is treated as a high-precision number - return 'H'; - } + return ubjson_integer_prefix(j.m_data.m_value.number_unsigned, use_bjdata); case value_t::number_float: - return get_ubjson_float_prefix(j.m_data.m_value.number_float); + return get_ubjson_float_prefix(); case value_t::string: return 'S'; @@ -22934,7 +22682,7 @@ class binary_writer Containers, strings, high-precision numbers, booleans and null cannot be declared as the single type of an optimized container in BJData; such a container is written unoptimized. The reader rejects them with the same - list (binary_reader::bjd_optimized_type_markers). + list (binary_reader::is_bjd_excluded_optimized_type()). */ static constexpr bool is_bjdata_excluded_type_marker(const CharType marker) noexcept { @@ -22942,14 +22690,16 @@ class binary_writer || marker == 'T' || marker == 'F' || marker == 'N' || marker == 'Z'; } - static constexpr CharType get_ubjson_float_prefix(float /*unused*/) + /// @return the UBJSON/BJData type marker for a float or double value + /// + /// number_float_t must be float or double; a static_assert (rather than + /// an ambiguous overload) reports an unsupported number_float_t clearly. + template + static constexpr CharType get_ubjson_float_prefix() { - return 'd'; // float 32 - } - - static constexpr CharType get_ubjson_float_prefix(double /*unused*/) - { - return 'D'; // float 64 + static_assert(std::is_same::value || std::is_same::value, + "number_float_t must be float or double for the UBJSON/BJData writer"); + return std::is_same::value ? 'd' : 'D'; // float 32 / float 64 } /*! @@ -22966,35 +22716,184 @@ class binary_writer : value_in_range_of(el.template get()); } + /*! + @brief look up the BJData ND-array dtype marker for an `_ArrayType_` name + @return the one-character marker, or '\0' if @a name does not name a known dtype + + A C++11 `constexpr` function cannot contain a `switch`, so this is a plain + comparison chain instead; it is only reached once per ND-array candidate + object. Keep in sync with binary_reader's `bjd_type_name()`, which maps + the other way. + */ + static CharType bjdata_ndarray_type_marker(const string_t& name) + { + if (name == "uint8") + { + return 'U'; + } + if (name == "int8") + { + return 'i'; + } + if (name == "uint16") + { + return 'u'; + } + if (name == "int16") + { + return 'I'; + } + if (name == "uint32") + { + return 'm'; + } + if (name == "int32") + { + return 'l'; + } + if (name == "uint64") + { + return 'M'; + } + if (name == "int64") + { + return 'L'; + } + if (name == "single") + { + return 'd'; + } + if (name == "double") + { + return 'D'; + } + if (name == "char") + { + return 'C'; + } + if (name == "byte") + { + return 'B'; + } + return '\0'; + } + + /*! + @brief validate (dry_run) or write one BJData ND-array element of integer dtype @a T + @return whether @a el's value is in range of @a T; always true when @a dry_run is false + */ + template + bool write_bjdata_ndarray_element(const BasicJsonType& el, const bool dry_run) + { + if (dry_run) + { + return bjdata_ndarray_value_in_range(el); + } + using storage_type = typename std::conditional::value, std::uint64_t, std::int64_t>::type; + write_number(static_cast(el.template get()), true); + return true; + } + + /*! + @brief validate (dry_run) or write one BJData ND-array element of dtype 'd' (single precision) + @return whether @a el's value fits a float without overflow; always true when @a dry_run is false + */ + bool write_bjdata_ndarray_float_element(const BasicJsonType& el, const bool dry_run) + { + const auto dval = el.template get(); + if (dry_run) + { + return !std::isfinite(dval) || + (dval >= static_cast(std::numeric_limits::lowest()) && + dval <= static_cast((std::numeric_limits::max)())); + } + write_number(static_cast(dval), true); + return true; + } + + /*! + @brief validate or write every element of a BJData ND-array's `_ArrayData_` + @param[in] array_data the `_ArrayData_` array + @param[in] dtype the ND-array dtype marker, as returned by bjdata_ndarray_type_marker() + @param[in] dry_run true to only range-check each element, false to write it + @return whether every element is in range for @a dtype (always true when @a dry_run is false) + */ + bool write_bjdata_ndarray_elements(const BasicJsonType& array_data, const CharType dtype, const bool dry_run) + { + for (const auto& el : array_data) + { + bool ok = true; + switch (dtype) + { + case 'U': + case 'C': + case 'B': + ok = write_bjdata_ndarray_element(el, dry_run); + break; + case 'i': + ok = write_bjdata_ndarray_element(el, dry_run); + break; + case 'u': + ok = write_bjdata_ndarray_element(el, dry_run); + break; + case 'I': + ok = write_bjdata_ndarray_element(el, dry_run); + break; + case 'm': + ok = write_bjdata_ndarray_element(el, dry_run); + break; + case 'l': + ok = write_bjdata_ndarray_element(el, dry_run); + break; + case 'M': + ok = write_bjdata_ndarray_element(el, dry_run); + break; + case 'L': + ok = write_bjdata_ndarray_element(el, dry_run); + break; + case 'd': + ok = write_bjdata_ndarray_float_element(el, dry_run); + break; + case 'D': + default: + // 'D' (double) already spans the full range of number_float_t + if (!dry_run) + { + write_number(el.template get(), true); + } + break; + } + if (!ok) + { + return false; + } + } + return true; + } + /*! @return false if the object is successfully converted to a bjdata ndarray, true if the type or size is invalid */ bool write_bjdata_ndarray(const typename BasicJsonType::object_t& value, const bool use_count, const bool use_type, const bjdata_version_t bjdata_version) { - std::map bjdtype = {{"uint8", 'U'}, {"int8", 'i'}, {"uint16", 'u'}, {"int16", 'I'}, - {"uint32", 'm'}, {"int32", 'l'}, {"uint64", 'M'}, {"int64", 'L'}, {"single", 'd'}, {"double", 'D'}, - {"char", 'C'}, {"byte", 'B'} - }; - - string_t key = "_ArrayType_"; + const auto& array_type = value.at("_ArrayType_"); // the type name is looked up as a string below; a non-string // annotation (e.g. a number, null, or an array) cannot name a known // dtype, so it is treated the same as an unrecognized type name and // falls back to a plain object encoding instead of throwing // type_error.302 out of get() - if (!value.at(key).is_string()) + if (!array_type.is_string()) { return true; } // use get() instead of static_cast to avoid an // ambiguous conversion under explicit instantiation on C++17 (see #4825) - auto it = bjdtype.find(value.at(key).template get()); - if (it == bjdtype.end()) + const CharType dtype = bjdata_ndarray_type_marker(array_type.template get()); + if (dtype == '\0') { return true; } - CharType dtype = it->second; // the 'B' (byte) marker is only defined from BJData Draft 3 onward; // emitting it under an earlier draft would produce a stream that an @@ -23006,12 +22905,12 @@ class binary_writer return true; } - key = "_ArraySize_"; + const auto& array_size = value.at("_ArraySize_"); // the dimensions are written verbatim as the header length below, so a // value that is not an array cannot produce a valid one: null emits 'Z' // and an object emits '{', neither of which a reader accepts after '#'. // Such an object is not a valid ndarray and falls back to a plain object. - if (!value.at(key).is_array()) + if (!array_size.is_array()) { return true; } @@ -23021,7 +22920,7 @@ class binary_writer // dimension, or a 1xN row vector is read back as a plain array, which // would silently drop the annotation, so such an object falls back to // a plain object encoding instead - const auto& dims = value.at(key); + const auto& dims = array_size; if (dims.size() < 2 || (dims.size() == 2 && dims.at(0).is_number_integer() && dims.at(0).template get() == 1)) { return true; @@ -23068,8 +22967,8 @@ class binary_writer // to be an array: size() is 0 for null and 1 for any other scalar, and // iterating an object visits its values, so any of these could match // the dimensions by accident and be encoded as an unrelated ND-array - key = "_ArrayData_"; - if (!value.at(key).is_array() || value.at(key).size() != len) + const auto& array_data = value.at("_ArrayData_"); + if (!array_data.is_array() || array_data.size() != len) { return true; } @@ -23084,7 +22983,7 @@ class binary_writer // API stores int literals as signed), so both are accepted here and the // writes below go through get<>, which reads the member that is active. const bool ndarray_is_float = (dtype == 'd' || dtype == 'D'); - for (const auto& el : value.at(key)) + for (const auto& el : array_data) { if (ndarray_is_float ? !el.is_number_float() : !el.is_number_integer()) { @@ -23097,134 +22996,19 @@ class binary_writer // wrap (integers) or overflow to infinity (the "single" precision // float) instead of being reported, so such an object falls back to // a plain object encoding as well - for (const auto& el : value.at(key)) + if (!write_bjdata_ndarray_elements(array_data, dtype, true)) { - bool in_range = true; - switch (dtype) - { - case 'U': - case 'C': - case 'B': - in_range = bjdata_ndarray_value_in_range(el); - break; - case 'i': - in_range = bjdata_ndarray_value_in_range(el); - break; - case 'u': - in_range = bjdata_ndarray_value_in_range(el); - break; - case 'I': - in_range = bjdata_ndarray_value_in_range(el); - break; - case 'm': - in_range = bjdata_ndarray_value_in_range(el); - break; - case 'l': - in_range = bjdata_ndarray_value_in_range(el); - break; - case 'M': - in_range = bjdata_ndarray_value_in_range(el); - break; - case 'L': - in_range = bjdata_ndarray_value_in_range(el); - break; - case 'd': - { - const auto dval = el.template get(); - in_range = !std::isfinite(dval) || - (dval >= static_cast(std::numeric_limits::lowest()) && - dval <= static_cast((std::numeric_limits::max)())); - break; - } - default: - // 'D' (double) already spans the full range of number_float_t - break; - } - if (!in_range) - { - return true; - } + return true; } - oa.write_character('['); - oa.write_character('$'); + oa.write_character(to_char_type('[')); + oa.write_character(to_char_type('$')); oa.write_character(dtype); - oa.write_character('#'); + oa.write_character(to_char_type('#')); - key = "_ArraySize_"; - write_ubjson(value.at(key), use_count, use_type, true, true, bjdata_version); + write_ubjson(array_size, use_count, use_type, true, true, bjdata_version); - key = "_ArrayData_"; - if (dtype == 'U' || dtype == 'C' || dtype == 'B') - { - for (const auto& el : value.at(key)) - { - write_number(static_cast(el.template get()), true); - } - } - else if (dtype == 'i') - { - for (const auto& el : value.at(key)) - { - write_number(static_cast(el.template get()), true); - } - } - else if (dtype == 'u') - { - for (const auto& el : value.at(key)) - { - write_number(static_cast(el.template get()), true); - } - } - else if (dtype == 'I') - { - for (const auto& el : value.at(key)) - { - write_number(static_cast(el.template get()), true); - } - } - else if (dtype == 'm') - { - for (const auto& el : value.at(key)) - { - write_number(static_cast(el.template get()), true); - } - } - else if (dtype == 'l') - { - for (const auto& el : value.at(key)) - { - write_number(static_cast(el.template get()), true); - } - } - else if (dtype == 'M') - { - for (const auto& el : value.at(key)) - { - write_number(el.template get(), true); - } - } - else if (dtype == 'L') - { - for (const auto& el : value.at(key)) - { - write_number(el.template get(), true); - } - } - else if (dtype == 'd') - { - for (const auto& el : value.at(key)) - { - write_number(static_cast(el.template get()), true); - } - } - else if (dtype == 'D') - { - for (const auto& el : value.at(key)) - { - write_number(el.template get(), true); - } - } + write_bjdata_ndarray_elements(array_data, dtype, false); return false; } @@ -23537,7 +23321,7 @@ class binary_writer } else { - write_compact_float(n, detail::input_format_t::bon8); + write_compact_float(n, to_char_type(0x8E), to_char_type(0x8F)); } #ifdef __GNUC__ JSON_HEDLEY_DIAGNOSTIC_POP @@ -23548,19 +23332,6 @@ class binary_writer // Utility functions // /////////////////////// - /* - @brief write a number to output input - @param[in] n number of type @a NumberType - @param[in] OutputIsLittleEndian Set to true if output data is - required to be little endian - @tparam NumberType the type of the number - - @note This function needs to respect the system's endianness, because bytes - in CBOR, MessagePack, and UBJSON are stored in network order (big - endian) and therefore need reordering on little endian systems. - On the other hand, BSON and BJData use little endian and should reorder - on big endian systems. - */ // single-instruction byte swaps (compilers lower these to bswap/rev/movbe); // used to emit big-endian numbers without a per-byte std::reverse loop static std::uint16_t byte_swap(std::uint16_t x) noexcept @@ -23642,6 +23413,19 @@ class binary_writer std::reverse(a.begin(), a.end()); } + /*! + @brief write a number to the output + @param[in] n number of type @a NumberType + @param[in] OutputIsLittleEndian Set to true if output data is + required to be little endian + @tparam NumberType the type of the number + + @note This function needs to respect the system's endianness, because bytes + in CBOR, MessagePack, UBJSON, and BON8 are stored in network order + (big endian) and therefore need reordering on little endian systems. + On the other hand, BSON and BJData use little endian and should + reorder on big endian systems. + */ template void write_number(const NumberType n, const bool OutputIsLittleEndian = false) { @@ -23659,8 +23443,17 @@ class binary_writer oa.write_characters(vec.data(), sizeof(NumberType)); } - void write_compact_float(const number_float_t n, detail::input_format_t format) + /// @brief write @a n using @a float32_marker if it round-trips through + /// float, otherwise using @a float64_marker + /// + /// @a float32_marker and @a float64_marker are the format-specific type + /// markers (CBOR: 0xFA/0xFB, MessagePack: 0xCA/0xCB, BON8: 0x8E/0x8F); + /// each caller already knows them at compile time, so the format itself + /// no longer needs to be passed in. + void write_compact_float(const number_float_t n, const CharType float32_marker, const CharType float64_marker) { + static_assert(std::is_same::value || std::is_same::value, + "number_float_t must be float or double for the CBOR/MessagePack/BON8 writer"); #ifdef __GNUC__ JSON_HEDLEY_DIAGNOSTIC_PUSH JSON_HEDLEY_PRAGMA(GCC diagnostic ignored "-Wfloat-equal") @@ -23678,12 +23471,12 @@ class binary_writer static_cast(n) <= static_cast((std::numeric_limits::max)()) && static_cast(static_cast(n)) == static_cast(n)))) { - oa.write_character(get_compact_float_prefix(format)); + oa.write_character(float32_marker); write_number(static_cast(n)); } else { - oa.write_character(get_compact_float_prefix(format)); + oa.write_character(float64_marker); write_number(n); } #ifdef __GNUC__ @@ -23692,7 +23485,7 @@ class binary_writer } public: - // The following to_char_type functions are implement the conversion + // The following to_char_type functions implement the conversion // between uint8_t and CharType. In case CharType is not unsigned, // such a conversion is required to allow values greater than 128. // See for a discussion. @@ -32168,7 +31961,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec auto ia = detail::input_adapter(std::forward(i)); return format == input_format_t::json ? parser(std::move(ia), nullptr, true, ignore_comments, ignore_trailing_commas).sax_parse(sax, strict) - : detail::binary_reader(std::move(ia), format).sax_parse(format, sax, strict); + : detail::binary_reader(std::move(ia), format).sax_parse(sax, strict); } /// @brief generate SAX events (iterator pair, or iterator+sentinel pair for C++20 ranges support) @@ -32185,7 +31978,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec auto ia = detail::input_adapter(std::move(first), std::move(last)); return format == input_format_t::json ? parser(std::move(ia), nullptr, true, ignore_comments, ignore_trailing_commas).sax_parse(sax, strict) - : detail::binary_reader(std::move(ia), format).sax_parse(format, sax, strict); + : detail::binary_reader(std::move(ia), format).sax_parse(sax, strict); } /// @brief generate SAX events @@ -32207,7 +32000,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg) ? parser(std::move(ia), nullptr, true, ignore_comments, ignore_trailing_commas).sax_parse(sax, strict) // NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg) - : detail::binary_reader(std::move(ia), format).sax_parse(format, sax, strict); + : detail::binary_reader(std::move(ia), format).sax_parse(sax, strict); } #ifndef JSON_NO_IO /// @brief deserialize from stream @@ -32505,7 +32298,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::forward(i)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::cbor).sax_parse(input_format_t::cbor, &sdp, strict, tag_handler)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::cbor).sax_parse(&sdp, strict, tag_handler)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } @@ -32525,7 +32318,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::move(first), std::move(last)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::cbor).sax_parse(input_format_t::cbor, &sdp, strict, tag_handler)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::cbor).sax_parse(&sdp, strict, tag_handler)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } @@ -32554,7 +32347,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec auto ia = i.get(); detail::json_sax_dom_parser sdp(result, allow_exceptions); // NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg) - if (!binary_reader(std::move(ia), input_format_t::cbor).sax_parse(input_format_t::cbor, &sdp, strict, tag_handler)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::cbor).sax_parse(&sdp, strict, tag_handler)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } @@ -32572,7 +32365,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::forward(i)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::msgpack).sax_parse(input_format_t::msgpack, &sdp, strict)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::msgpack).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } @@ -32591,7 +32384,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::move(first), std::move(last)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::msgpack).sax_parse(input_format_t::msgpack, &sdp, strict)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::msgpack).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } @@ -32618,7 +32411,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec auto ia = i.get(); detail::json_sax_dom_parser sdp(result, allow_exceptions); // NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg) - if (!binary_reader(std::move(ia), input_format_t::msgpack).sax_parse(input_format_t::msgpack, &sdp, strict)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::msgpack).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } @@ -32636,7 +32429,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::forward(i)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::ubjson).sax_parse(input_format_t::ubjson, &sdp, strict)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::ubjson).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } @@ -32655,7 +32448,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::move(first), std::move(last)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::ubjson).sax_parse(input_format_t::ubjson, &sdp, strict)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::ubjson).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } @@ -32682,7 +32475,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec auto ia = i.get(); detail::json_sax_dom_parser sdp(result, allow_exceptions); // NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg) - if (!binary_reader(std::move(ia), input_format_t::ubjson).sax_parse(input_format_t::ubjson, &sdp, strict)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::ubjson).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } @@ -32700,7 +32493,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::forward(i)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::bjdata).sax_parse(input_format_t::bjdata, &sdp, strict)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::bjdata).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } @@ -32719,7 +32512,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::move(first), std::move(last)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::bjdata).sax_parse(input_format_t::bjdata, &sdp, strict)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::bjdata).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } @@ -32737,7 +32530,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::forward(i)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::bon8).sax_parse(input_format_t::bon8, &sdp, strict)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::bon8).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } @@ -32756,7 +32549,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::move(first), std::move(last)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::bon8).sax_parse(input_format_t::bon8, &sdp, strict)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::bon8).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } @@ -32774,7 +32567,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::forward(i)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::bson).sax_parse(input_format_t::bson, &sdp, strict)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::bson).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } @@ -32793,7 +32586,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec basic_json result; auto ia = detail::input_adapter(std::move(first), std::move(last)); detail::json_sax_dom_parser sdp(result, allow_exceptions); - if (!binary_reader(std::move(ia), input_format_t::bson).sax_parse(input_format_t::bson, &sdp, strict)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::bson).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } @@ -32820,7 +32613,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec auto ia = i.get(); detail::json_sax_dom_parser sdp(result, allow_exceptions); // NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg) - if (!binary_reader(std::move(ia), input_format_t::bson).sax_parse(input_format_t::bson, &sdp, strict)) // cppcheck-suppress[accessMoved] + if (!binary_reader(std::move(ia), input_format_t::bson).sax_parse(&sdp, strict)) // cppcheck-suppress[accessMoved] { result = value_t::discarded; } diff --git a/tests/src/unit-bjdata.cpp b/tests/src/unit-bjdata.cpp index a58507c15..c56d09e75 100644 --- a/tests/src/unit-bjdata.cpp +++ b/tests/src/unit-bjdata.cpp @@ -210,15 +210,42 @@ TEST_CASE_TEMPLATE_INVOKE(value_in_range_of_test, \ TEST_CASE("BJData") { - SECTION("binary_reader BJData LUT arrays are sorted") + SECTION("binary_reader BJData lookup tables") { std::vector const data; auto ia = nlohmann::detail::input_adapter(data); // NOLINTNEXTLINE(hicpp-move-const-arg,performance-move-const-arg) nlohmann::detail::binary_reader const br{std::move(ia), json::input_format_t::bjdata}; - CHECK(std::is_sorted(br.bjd_optimized_type_markers.begin(), br.bjd_optimized_type_markers.end())); - CHECK(std::is_sorted(br.bjd_types_map.begin(), br.bjd_types_map.end())); + // the excluded optimized-type markers must match binary_writer's + // is_bjdata_excluded_type_marker(), which encodes the same 8 markers + for (const char marker : + {'[', '{', 'S', 'H', 'T', 'F', 'N', 'Z' + }) + { + CHECK(br.is_bjd_excluded_optimized_type(marker)); + } + for (const char marker : + {'U', 'i', 'u', 'I', 'm', 'l', 'M', 'L', 'd', 'D', 'C', 'B', 'x' + }) + { + CHECK(!br.is_bjd_excluded_optimized_type(marker)); + } + + // every dtype marker must round-trip to its ND-array type name + const std::vector> types + { + {'B', "byte"}, {'C', "char"}, {'D', "double"}, {'I', "int16"}, + {'L', "int64"}, {'M', "uint64"}, {'U', "uint8"}, {'d', "single"}, + {'i', "int8"}, {'l', "int32"}, {'m', "uint32"}, {'u', "uint16"} + }; + for (const auto& type : types) + { + const char* name = br.bjd_type_name(type.first); + REQUIRE(name != nullptr); + CHECK(std::string(name) == type.second); + } + CHECK(br.bjd_type_name('x') == nullptr); } SECTION("individual values") diff --git a/tests/src/unit-ubjson.cpp b/tests/src/unit-ubjson.cpp index c315fac94..80c771d42 100644 --- a/tests/src/unit-ubjson.cpp +++ b/tests/src/unit-ubjson.cpp @@ -3033,3 +3033,224 @@ TEST_CASE("UBJSON optimized array of unsigned integers beyond int64") CHECK(json::to_ubjson(j, true, true) == expected); CHECK(json::from_ubjson(expected) == j); } + +namespace +{ +// the bytes that follow the marker of an integer: the value in the width of +// the marker (big endian for UBJSON, little endian for BJData), or, for a +// high-precision number, the length and the decimal digits +std::vector integer_payload(const char marker, const json& value, const bool little_endian) +{ + std::size_t width = 0; + switch (marker) + { + case 'i': + case 'U': + width = 1; + break; + case 'I': + case 'u': + width = 2; + break; + case 'l': + case 'm': + width = 4; + break; + case 'L': + case 'M': + width = 8; + break; + default: + { + const std::string digits = value.dump(); + std::vector result = {'i', static_cast(digits.size())}; + for (const char c : digits) + { + result.push_back(static_cast(c)); + } + return result; + } + } + + const std::uint64_t bits = value.is_number_unsigned() + ? value.get() + : static_cast(value.get()); + std::vector result(width); + for (std::size_t i = 0; i < width; ++i) + { + result[little_endian ? i : width - 1 - i] = static_cast(bits >> (8 * i)); + } + return result; +} + +json i64(const std::int64_t v) +{ + return v; +} + +json u64(const std::uint64_t v) +{ + return v; +} +} // namespace + +TEST_CASE("UBJSON and BJData integer markers at every range edge") +{ + // An optimized container announces the marker of its values after `$` and + // then writes every value without a marker, so the marker the writer + // announces and the width it writes must match for every value. This + // checks both for the values around each edge of the integer types, as + // scalars and as the values of optimized arrays and objects. + struct integer_case + { + json value; + char ubjson; // expected UBJSON marker + char bjdata; // expected BJData marker + }; + + const std::int64_t int64_min = (std::numeric_limits::min)(); + const std::int64_t int64_max = (std::numeric_limits::max)(); + const std::uint64_t uint64_max = (std::numeric_limits::max)(); + + const std::vector cases = + { + // int8 + {i64(-129), 'I', 'I'}, + {i64(-128), 'i', 'i'}, + {i64(-127), 'i', 'i'}, + {i64(-1), 'i', 'i'}, + {i64(0), 'i', 'i'}, + {u64(0), 'i', 'i'}, + {i64(126), 'i', 'i'}, + {i64(127), 'i', 'i'}, + {u64(127), 'i', 'i'}, + {i64(128), 'U', 'U'}, + {u64(128), 'U', 'U'}, + // uint8 + {i64(254), 'U', 'U'}, + {i64(255), 'U', 'U'}, + {u64(255), 'U', 'U'}, + {i64(256), 'I', 'I'}, + {u64(256), 'I', 'I'}, + // int16 + {i64(-32769), 'l', 'l'}, + {i64(-32768), 'I', 'I'}, + {i64(-32767), 'I', 'I'}, + {i64(32766), 'I', 'I'}, + {i64(32767), 'I', 'I'}, + {u64(32767), 'I', 'I'}, + {i64(32768), 'l', 'u'}, + {u64(32768), 'l', 'u'}, + // uint16 (BJData only) + {i64(65534), 'l', 'u'}, + {i64(65535), 'l', 'u'}, + {u64(65535), 'l', 'u'}, + {i64(65536), 'l', 'l'}, + {u64(65536), 'l', 'l'}, + // int32 + {i64(-2147483649LL), 'L', 'L'}, + {i64(-2147483648LL), 'l', 'l'}, + {i64(-2147483647LL), 'l', 'l'}, + {i64(2147483646LL), 'l', 'l'}, + {i64(2147483647LL), 'l', 'l'}, + {u64(2147483647ULL), 'l', 'l'}, + {i64(2147483648LL), 'L', 'm'}, + {u64(2147483648ULL), 'L', 'm'}, + // uint32 (BJData only) + {i64(4294967294LL), 'L', 'm'}, + {i64(4294967295LL), 'L', 'm'}, + {u64(4294967295ULL), 'L', 'm'}, + {i64(4294967296LL), 'L', 'L'}, + {u64(4294967296ULL), 'L', 'L'}, + // int64 + {i64(int64_min), 'L', 'L'}, + {i64(int64_min + 1), 'L', 'L'}, + {i64(int64_max - 1), 'L', 'L'}, + {i64(int64_max), 'L', 'L'}, + {u64(static_cast(int64_max)), 'L', 'L'}, + // uint64 (BJData only; UBJSON writes a high-precision number) + {u64(static_cast(int64_max) + 1), 'H', 'M'}, + {u64(uint64_max - 1), 'H', 'M'}, + {u64(uint64_max), 'H', 'M'}, + }; + + for (const auto& c : cases) + { + for (const bool bjdata : + { + false, true + }) + { + const char marker = bjdata ? c.bjdata : c.ubjson; + const std::vector payload = integer_payload(marker, c.value, bjdata); + const auto to_binary = [bjdata](const json & j, const bool use_size, const bool use_type) + { + return bjdata ? json::to_bjdata(j, use_size, use_type) : json::to_ubjson(j, use_size, use_type); + }; + const auto from_binary = [bjdata](const std::vector& v) + { + return bjdata ? json::from_bjdata(v) : json::from_ubjson(v); + }; + INFO("value = " << c.value.dump() << (c.value.is_number_unsigned() ? " (unsigned)" : "") << ", format = " << (bjdata ? "BJData" : "UBJSON")); + + // scalar + std::vector expected = {static_cast(marker)}; + expected.insert(expected.end(), payload.begin(), payload.end()); + for (const bool use_size : + { + false, true + }) + { + CHECK(to_binary(c.value, use_size, false) == expected); + } + CHECK(from_binary(expected) == c.value); + + const json arr = {c.value, c.value, c.value}; + + // array without count or type: every value has its marker + expected = {'['}; + for (int i = 0; i < 3; ++i) + { + expected.push_back(static_cast(marker)); + expected.insert(expected.end(), payload.begin(), payload.end()); + } + expected.push_back(']'); + CHECK(to_binary(arr, false, false) == expected); + CHECK(from_binary(expected) == arr); + + // array with count: every value has its marker + expected = {'[', '#', 'i', 3}; + for (int i = 0; i < 3; ++i) + { + expected.push_back(static_cast(marker)); + expected.insert(expected.end(), payload.begin(), payload.end()); + } + CHECK(to_binary(arr, true, false) == expected); + CHECK(from_binary(expected) == arr); + + // array with type and count: the marker once, then the payloads + expected = {'[', '$', static_cast(marker), '#', 'i', 3}; + for (int i = 0; i < 3; ++i) + { + expected.insert(expected.end(), payload.begin(), payload.end()); + } + CHECK(to_binary(arr, true, true) == expected); + CHECK(from_binary(expected) == arr); + + // object with type and count: the marker once, then key and payload + const json obj = {{"a", c.value}, {"b", c.value}}; + expected = {'{', '$', static_cast(marker), '#', 'i', 2}; + for (const char key : + {'a', 'b' + }) + { + expected.push_back('i'); + expected.push_back(1); + expected.push_back(static_cast(key)); + expected.insert(expected.end(), payload.begin(), payload.end()); + } + CHECK(to_binary(obj, true, true) == expected); + CHECK(from_binary(expected) == obj); + } + } +}