From f3768d686841e0417f7ca09f0d98d58e08ad7e0b Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Thu, 24 Sep 2026 17:01:19 +0200 Subject: [PATCH 01/17] Match ABI tag order in namespace tests to abi_macros.hpp (#5551) * Match ABI tag order in namespace tests to abi_macros.hpp NLOHMANN_JSON_ABI_TAGS concatenates the tags as _diag, _ldvcmp, _dp, but the default and noversion ABI tests expected _diag, _dp, _ldvcmp. The tests therefore failed whenever both JSON_DIAGNOSTIC_POSITIONS and JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON were enabled, a combination CI never exercises. Reorder the expectations to match the header. Also document the _dp tag in the namespace feature page, which listed only _diag and _ldvcmp. Signed-off-by: Niels Lohmann * Test the ABI namespace with all ABI tags enabled Build the default and noversion ABI config tests a second time with JSON_DIAGNOSTICS, JSON_DIAGNOSTIC_POSITIONS and JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON all set, so the expected tag order is checked on every test run instead of depending on which CMake options a CI job happens to enable. Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- docs/mkdocs/docs/features/namespace.md | 1 + tests/abi/config/CMakeLists.txt | 14 ++++++++++++++ tests/abi/config/default.cpp | 8 ++++---- tests/abi/config/noversion.cpp | 8 ++++---- 4 files changed, 23 insertions(+), 8 deletions(-) diff --git a/docs/mkdocs/docs/features/namespace.md b/docs/mkdocs/docs/features/namespace.md index 5542c1f88..c4efe772a 100644 --- a/docs/mkdocs/docs/features/namespace.md +++ b/docs/mkdocs/docs/features/namespace.md @@ -15,6 +15,7 @@ The complete default namespace name is derived as follows: - [`JSON_DIAGNOSTICS`](../api/macros/json_diagnostics.md) defined non-zero appends `_diag`. - [`JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON`](../api/macros/json_use_legacy_discarded_value_comparison.md) defined non-zero appends `_ldvcmp`. + - [`JSON_DIAGNOSTIC_POSITIONS`](../api/macros/json_diagnostic_positions.md) defined non-zero appends `_dp`. - The inline namespace ends with the suffix `_v` followed by the 3 components of the version number separated by underscores. To omit the version component, see [Disabling the version component](#disabling-the-version-component) below. diff --git a/tests/abi/config/CMakeLists.txt b/tests/abi/config/CMakeLists.txt index 3a8367690..52941dc33 100644 --- a/tests/abi/config/CMakeLists.txt +++ b/tests/abi/config/CMakeLists.txt @@ -14,6 +14,20 @@ add_test( NAME test-abi_config_noversion COMMAND abi_config_noversion ${DOCTEST_TEST_FILTER}) +# test default and no version namespace with all ABI tags enabled, so the +# expected tag order is checked regardless of the JSON_* CMake options +foreach(test default noversion) + add_executable(abi_config_${test}_all_tags ${test}.cpp) + target_compile_definitions(abi_config_${test}_all_tags PRIVATE + JSON_DIAGNOSTICS=1 + JSON_DIAGNOSTIC_POSITIONS=1 + JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON=1) + target_link_libraries(abi_config_${test}_all_tags PRIVATE abi_compat_main) + add_test( + NAME test-abi_config_${test}_all_tags + COMMAND abi_config_${test}_all_tags ${DOCTEST_TEST_FILTER}) +endforeach() + # test custom namespace add_executable(abi_config_custom custom.cpp) target_link_libraries(abi_config_custom PRIVATE abi_compat_main) diff --git a/tests/abi/config/default.cpp b/tests/abi/config/default.cpp index 0edc12e62..f3ee23110 100644 --- a/tests/abi/config/default.cpp +++ b/tests/abi/config/default.cpp @@ -24,14 +24,14 @@ TEST_CASE("default namespace") expected += "_diag"; #endif -#if JSON_DIAGNOSTIC_POSITIONS - expected += "_dp"; -#endif - #if JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON expected += "_ldvcmp"; #endif +#if JSON_DIAGNOSTIC_POSITIONS + expected += "_dp"; +#endif + expected += "_v" STRINGIZE(NLOHMANN_JSON_VERSION_MAJOR); expected += "_" STRINGIZE(NLOHMANN_JSON_VERSION_MINOR); expected += "_" STRINGIZE(NLOHMANN_JSON_VERSION_PATCH) "::basic_json"; diff --git a/tests/abi/config/noversion.cpp b/tests/abi/config/noversion.cpp index 2ae5cf5ac..cbdcb149b 100644 --- a/tests/abi/config/noversion.cpp +++ b/tests/abi/config/noversion.cpp @@ -25,14 +25,14 @@ TEST_CASE("default namespace without version component") expected += "_diag"; #endif -#if JSON_DIAGNOSTIC_POSITIONS - expected += "_dp"; -#endif - #if JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON expected += "_ldvcmp"; #endif +#if JSON_DIAGNOSTIC_POSITIONS + expected += "_dp"; +#endif + expected += "::basic_json"; // fallback for Clang From 918da646577fd1a004cd72efbc7cbce2f578783b Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Thu, 24 Sep 2026 17:01:31 +0200 Subject: [PATCH 02/17] Keep BJData ndarray annotations that would not survive a round trip as objects (#5542) write_bjdata_ndarray() encoded a JData-annotated object as a BJData ND-array whenever its dimensions' product matched _ArrayData_.size(), which lost information in two ways: - _ArrayData_ was never required to be an array. null has size 0, any other scalar has size 1, and iterating an object visits its values, so e.g. {"_ArraySize_":[1],"_ArrayData_":5} was written as the array [5], and an object _ArrayData_ came back as an array. - The reader only restores an annotated object from an ND-array with at least two non-zero dimensions that is not a 1xN row vector; an empty, 1-D, row-vector, or zero-sized shape is read back as a plain array. The writer nonetheless emitted ND-array headers for these shapes, so the annotation was silently dropped. OSS-Fuzz issue 563659413 hit this in parse_bjdata_fuzzer: an empty binary _ArraySize_ is written as a plain object and read back as an empty array, after which {"_ArrayType_":"int16","_ArraySize_":[],"_ArrayData_":null} was encoded as the ND-array header "[$I#[]" and re-read as [], failing the harness's value-stability check. Such objects now fall back to a plain object encoding, which round-trips. Genuine ND-arrays (two or more positive dimensions, not a 1xN row vector) are encoded exactly as before. Existing fallback tests that used 1-D shapes are moved to 2-D shapes so they keep exercising the check they were written for, and the BJData documentation is updated. Signed-off-by: Niels Lohmann --- .../docs/features/binary_formats/bjdata.md | 18 +-- .../nlohmann/detail/output/binary_writer.hpp | 30 ++++- single_include/nlohmann/json.hpp | 30 ++++- tests/src/unit-bjdata.cpp | 112 +++++++++++++++--- 4 files changed, 156 insertions(+), 34 deletions(-) diff --git a/docs/mkdocs/docs/features/binary_formats/bjdata.md b/docs/mkdocs/docs/features/binary_formats/bjdata.md index d3f63a9b8..4cb61053f 100644 --- a/docs/mkdocs/docs/features/binary_formats/bjdata.md +++ b/docs/mkdocs/docs/features/binary_formats/bjdata.md @@ -116,18 +116,22 @@ The library uses the following mapping from JSON values types to BJData types ac ``` Likewise, when a JSON object in the above form is serialized using - [`to_bjdata`](../../api/basic_json/to_bjdata.md), it is automatically converted into a compact BJData ND-array. When - the 1-dimensional vector stored in `"_ArraySize_"` contains a single integer or two integers with one being 1, a - regular 1-D optimized array is generated instead. + [`to_bjdata`](../../api/basic_json/to_bjdata.md), it is automatically converted into a compact BJData ND-array. - An object is only converted if the annotation actually describes a packed array; otherwise it is serialized as a - regular JSON object. This requires all of the following: + When parsing, an ND-array whose dimension vector is empty, contains a single integer, contains two integers with the + first being 1, or contains a 0 is returned as a regular (possibly empty) array rather than an annotated object. + + An object is only converted if the annotation describes a packed array that is parsed back into the same annotated + object; otherwise it is serialized as a regular JSON object, so the annotation is never lost in a round trip. This requires + all of the following: - `"_ArrayType_"` is one of `uint8`, `int8`, `uint16`, `int16`, `uint32`, `int32`, `uint64`, `int64`, `single`, `double`, `char`, or `byte`, - `"_ArraySize_"` is an array, since the dimensions are written as the ND-array header's length, - - every entry of `"_ArraySize_"` is a non-negative integer, and their product is representable as a `std::size_t`, - - `"_ArrayData_"` holds exactly that many elements, and + - `"_ArraySize_"` has at least two entries and is not a 1×N row vector (first entry 1), since other shapes are + parsed back as a regular array, + - every entry of `"_ArraySize_"` is a positive integer, and their product is representable as a `std::size_t`, + - `"_ArrayData_"` is an array holding exactly that many elements, and - every element of `"_ArrayData_"` is a number of the kind named by `"_ArrayType_"` (a floating-point number for `single` and `double`, an integer otherwise). diff --git a/include/nlohmann/detail/output/binary_writer.hpp b/include/nlohmann/detail/output/binary_writer.hpp index a355f1c15..495f872c0 100644 --- a/include/nlohmann/detail/output/binary_writer.hpp +++ b/include/nlohmann/detail/output/binary_writer.hpp @@ -1800,8 +1800,19 @@ class binary_writer return true; } - std::size_t len = (value.at(key).empty() ? 0 : 1); - for (const auto& el : value.at(key)) + // the reader only restores an annotated object from an ND-array header + // with at least two dimensions: an empty dimension vector, a single + // dimension, or a 1xN row vector is read back as a plain array, which + // would silently drop the annotation, so such an object falls back to + // a plain object encoding instead + const auto& dims = value.at(key); + if (dims.size() < 2 || (dims.size() == 2 && dims.at(0).is_number_integer() && dims.at(0).template get() == 1)) + { + return true; + } + + std::size_t len = 1; + for (const auto& el : dims) { // a dimension is read as an unsigned value below, so anything that // is not a non-negative integer is rejected: a non-integer entry @@ -1823,15 +1834,26 @@ class binary_writer return true; } const auto dim_size = static_cast(dim); - if (dim_size != 0 && len > (std::numeric_limits::max)() / dim_size) + + // the reader turns an ND-array with any zero dimension into an + // empty plain array, dropping the annotation, so keep the object + if (dim_size == 0) + { + return true; + } + if (len > (std::numeric_limits::max)() / dim_size) { return true; } len *= dim_size; } + // the elements are written from _ArrayData_ as a flat list, so it has + // to be an array: size() is 0 for null and 1 for any other scalar, and + // iterating an object visits its values, so any of these could match + // the dimensions by accident and be encoded as an unrelated ND-array key = "_ArrayData_"; - if (value.at(key).size() != len) + if (!value.at(key).is_array() || value.at(key).size() != len) { return true; } diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 81b0e09f6..1d5e21927 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -20664,8 +20664,19 @@ class binary_writer return true; } - std::size_t len = (value.at(key).empty() ? 0 : 1); - for (const auto& el : value.at(key)) + // the reader only restores an annotated object from an ND-array header + // with at least two dimensions: an empty dimension vector, a single + // dimension, or a 1xN row vector is read back as a plain array, which + // would silently drop the annotation, so such an object falls back to + // a plain object encoding instead + const auto& dims = value.at(key); + if (dims.size() < 2 || (dims.size() == 2 && dims.at(0).is_number_integer() && dims.at(0).template get() == 1)) + { + return true; + } + + std::size_t len = 1; + for (const auto& el : dims) { // a dimension is read as an unsigned value below, so anything that // is not a non-negative integer is rejected: a non-integer entry @@ -20687,15 +20698,26 @@ class binary_writer return true; } const auto dim_size = static_cast(dim); - if (dim_size != 0 && len > (std::numeric_limits::max)() / dim_size) + + // the reader turns an ND-array with any zero dimension into an + // empty plain array, dropping the annotation, so keep the object + if (dim_size == 0) + { + return true; + } + if (len > (std::numeric_limits::max)() / dim_size) { return true; } len *= dim_size; } + // the elements are written from _ArrayData_ as a flat list, so it has + // to be an array: size() is 0 for null and 1 for any other scalar, and + // iterating an object visits its values, so any of these could match + // the dimensions by accident and be encoded as an unrelated ND-array key = "_ArrayData_"; - if (value.at(key).size() != len) + if (!value.at(key).is_array() || value.at(key).size() != len) { return true; } diff --git a/tests/src/unit-bjdata.cpp b/tests/src/unit-bjdata.cpp index 334259fb7..9bba141d2 100644 --- a/tests/src/unit-bjdata.cpp +++ b/tests/src/unit-bjdata.cpp @@ -2604,25 +2604,25 @@ TEST_CASE("BJData") // that still round-trips. // string data declared as a uint64 array - json const j_str = json({{"_ArrayType_", "uint64"}, {"_ArraySize_", {1}}, {"_ArrayData_", {"pointer"}}}); + json const j_str = json({{"_ArrayType_", "uint64"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {"pointer", "value"}}}); const auto out_str = json::to_bjdata(j_str); CHECK(out_str.at(0) == '{'); CHECK(json::from_bjdata(out_str) == j_str); // integer data declared as a double array - json const j_float = json({{"_ArrayType_", "double"}, {"_ArraySize_", {2}}, {"_ArrayData_", {1, 2}}}); + json const j_float = json({{"_ArrayType_", "double"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {1, 2}}}); const auto out_float = json::to_bjdata(j_float); CHECK(out_float.at(0) == '{'); CHECK(json::from_bjdata(out_float) == j_float); // a non-integer shape entry is likewise not treated as an ndarray - json const j_size = json({{"_ArrayType_", "uint8"}, {"_ArraySize_", {"x"}}, {"_ArrayData_", {1}}}); + json const j_size = json({{"_ArrayType_", "uint8"}, {"_ArraySize_", {"x", 1}}, {"_ArrayData_", {1}}}); const auto out_size = json::to_bjdata(j_size); CHECK(out_size.at(0) == '{'); CHECK(json::from_bjdata(out_size) == j_size); // a negative shape entry is not a usable dimension either - json const j_neg = json::parse(R"({"_ArrayType_":"uint8","_ArraySize_":[-1],"_ArrayData_":[1]})"); + json const j_neg = json::parse(R"({"_ArrayType_":"uint8","_ArraySize_":[-1,1],"_ArrayData_":[1]})"); const auto out_neg = json::to_bjdata(j_neg); CHECK(out_neg.at(0) == '{'); CHECK(json::from_bjdata(out_neg) == j_neg); @@ -2657,14 +2657,14 @@ TEST_CASE("BJData") } // negative values under a signed type behave the same way - const auto from_neg = json::to_bjdata(json::parse(R"({"_ArrayType_":"int32","_ArraySize_":[2],"_ArrayData_":[-5,7]})")); + const auto from_neg = json::to_bjdata(json::parse(R"({"_ArrayType_":"int32","_ArraySize_":[2,1],"_ArrayData_":[-5,7]})")); CHECK(from_neg.at(0) == '['); - CHECK(from_neg == json::to_bjdata(json({{"_ArrayType_", "int32"}, {"_ArraySize_", {2}}, {"_ArrayData_", {-5, 7}}}))); + CHECK(from_neg == json::to_bjdata(json({{"_ArrayType_", "int32"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {-5, 7}}}))); // and so do the floating point types - const auto from_float = json::to_bjdata(json::parse(R"({"_ArrayType_":"double","_ArraySize_":[2],"_ArrayData_":[1.5,2.5]})")); + const auto from_float = json::to_bjdata(json::parse(R"({"_ArrayType_":"double","_ArraySize_":[2,1],"_ArrayData_":[1.5,2.5]})")); CHECK(from_float.at(0) == '['); - CHECK(from_float == json::to_bjdata(json({{"_ArrayType_", "double"}, {"_ArraySize_", {2}}, {"_ArrayData_", {1.5, 2.5}}}))); + CHECK(from_float == json::to_bjdata(json({{"_ArrayType_", "double"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {1.5, 2.5}}}))); } SECTION("optimized ndarray (type and vector-size as 1D array)") @@ -2835,7 +2835,7 @@ TEST_CASE("BJData") // a single dimension that does not fit into std::size_t is // rejected for the same reason (only observable where // std::size_t is narrower than 64 bit) - json j_huge = json({{"_ArrayData_", json::array()}, {"_ArraySize_", {18446744073709551615ull}}, {"_ArrayType_", "uint8"}}); + json j_huge = json({{"_ArrayData_", json::array()}, {"_ArraySize_", {18446744073709551615ull, 2}}, {"_ArrayType_", "uint8"}}); CHECK(json::from_bjdata(json::to_bjdata(j_huge), true, true) == j_huge); // a well-formed ndarray is still encoded as one @@ -2878,42 +2878,116 @@ TEST_CASE("BJData") // object encoding that still round-trips (see GitHub issue #5403) // an unsigned element that does not fit uint8 - json const j_uint8 = json({{"_ArrayType_", "uint8"}, {"_ArraySize_", {2}}, {"_ArrayData_", {1, 256}}}); + json const j_uint8 = json({{"_ArrayType_", "uint8"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {1, 256}}}); const auto out_uint8 = json::to_bjdata(j_uint8); CHECK(out_uint8.at(0) == '{'); CHECK(json::from_bjdata(out_uint8) == j_uint8); // a signed element that does not fit int8 - json const j_int8 = json({{"_ArrayType_", "int8"}, {"_ArraySize_", {2}}, {"_ArrayData_", {1, 200}}}); + json const j_int8 = json({{"_ArrayType_", "int8"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {1, 200}}}); const auto out_int8 = json::to_bjdata(j_int8); CHECK(out_int8.at(0) == '{'); CHECK(json::from_bjdata(out_int8) == j_int8); // a negative element is likewise out of range for an // unsigned _ArrayType_ - json const j_uint16_neg = json({{"_ArrayType_", "uint16"}, {"_ArraySize_", {2}}, {"_ArrayData_", {1, -1}}}); + json const j_uint16_neg = json({{"_ArrayType_", "uint16"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {1, -1}}}); const auto out_uint16_neg = json::to_bjdata(j_uint16_neg); CHECK(out_uint16_neg.at(0) == '{'); CHECK(json::from_bjdata(out_uint16_neg) == j_uint16_neg); // a double element that overflows to infinity when narrowed // to the "single" (float) precision named by _ArrayType_ - json const j_single = json({{"_ArrayType_", "single"}, {"_ArraySize_", {2}}, {"_ArrayData_", {1.5, 1e40}}}); + json const j_single = json({{"_ArrayType_", "single"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {1.5, 1e40}}}); const auto out_single = json::to_bjdata(j_single); CHECK(out_single.at(0) == '{'); CHECK(json::from_bjdata(out_single) == j_single); // in-range boundary values still use the compact ndarray encoding - json const j_uint8_ok = json({{"_ArrayType_", "uint8"}, {"_ArraySize_", {2}}, {"_ArrayData_", {0, 255}}}); - CHECK(json::to_bjdata(j_uint8_ok) == std::vector({'[', '$', 'U', '#', '[', 'i', 2, ']', 0, 255})); + json const j_uint8_ok = json({{"_ArrayType_", "uint8"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {0, 255}}}); + CHECK(json::to_bjdata(j_uint8_ok) == std::vector({'[', '$', 'U', '#', '[', 'i', 2, 'i', 1, ']', 0, 255})); - json const j_int8_ok = json({{"_ArrayType_", "int8"}, {"_ArraySize_", {2}}, {"_ArrayData_", {-128, 127}}}); - CHECK(json::to_bjdata(j_int8_ok) == std::vector({'[', '$', 'i', '#', '[', 'i', 2, ']', 0x80, 0x7F})); + json const j_int8_ok = json({{"_ArrayType_", "int8"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {-128, 127}}}); + CHECK(json::to_bjdata(j_int8_ok) == std::vector({'[', '$', 'i', '#', '[', 'i', 2, 'i', 1, ']', 0x80, 0x7F})); - json const j_single_ok = json({{"_ArrayType_", "single"}, {"_ArraySize_", {1}}, {"_ArrayData_", {1.5}}}); + json const j_single_ok = json({{"_ArrayType_", "single"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {1.5, -1.5}}}); const auto out_single_ok = json::to_bjdata(j_single_ok); CHECK(out_single_ok.at(0) == '['); - CHECK(json::from_bjdata(out_single_ok) == json({1.5f})); + CHECK(json::from_bjdata(out_single_ok) == json({{"_ArrayType_", "single"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {1.5f, -1.5f}}})); + } + + SECTION("ndarray that would not be read back as an annotated object stays as object") + { + // the reader only restores an annotated object from an ND-array + // with at least two non-zero dimensions that is not a 1xN row + // vector; any other shape is read back as a plain array. Writing + // such an object as an ND-array would drop its annotation, so it + // falls back to a plain object encoding that round-trips. + for (const char* text : + { + R"({"_ArrayType_":"int16","_ArraySize_":[],"_ArrayData_":[]})", + R"({"_ArrayType_":"int16","_ArraySize_":[2],"_ArrayData_":[1,2]})", + R"({"_ArrayType_":"int16","_ArraySize_":[1,2],"_ArrayData_":[1,2]})", + R"({"_ArrayType_":"int16","_ArraySize_":[0],"_ArrayData_":[]})", + R"({"_ArrayType_":"int16","_ArraySize_":[2,0],"_ArrayData_":[]})", + R"({"_ArrayType_":"int16","_ArraySize_":[0,2],"_ArrayData_":[]})" + }) + { + CAPTURE(text); + const json j = json::parse(text); + for (const bool use_size : + { + false, true + }) + { + const auto out = json::to_bjdata(j, use_size, use_size); + CHECK(out.at(0) == '{'); + CHECK(json::from_bjdata(out) == j); + } + } + + // a genuine ND-array still uses the compact encoding and round-trips + const json j_2d = json::parse(R"({"_ArrayType_":"int16","_ArraySize_":[2,1],"_ArrayData_":[1,2]})"); + const auto out_2d = json::to_bjdata(j_2d); + CHECK(out_2d.at(0) == '['); + CHECK(json::from_bjdata(out_2d) == j_2d); + } + + SECTION("ndarray with non-array _ArrayData_ stays as object") + { + // the elements are written from _ArrayData_ as a flat list, so it + // has to be an array: null has size 0, any other scalar has size 1, + // and iterating an object visits its values, so each of these could + // match the dimensions and be encoded as an unrelated ND-array + for (const char* text : + { + R"({"_ArrayType_":"int16","_ArraySize_":[2,1],"_ArrayData_":null})", + R"({"_ArrayType_":"int16","_ArraySize_":[2,1],"_ArrayData_":{"a":1,"b":2}})", + R"({"_ArrayType_":"int16","_ArraySize_":[1],"_ArrayData_":5})", + R"({"_ArrayType_":"int16","_ArraySize_":[],"_ArrayData_":null})" + }) + { + CAPTURE(text); + const json j = json::parse(text); + const auto out = json::to_bjdata(j); + CHECK(out.at(0) == '{'); + CHECK(json::from_bjdata(out) == j); + } + + // OSS-Fuzz issue 563659413: an empty binary _ArraySize_ is written + // as a plain object and read back as an empty array, after which + // the object with a null _ArrayData_ was encoded as an empty + // ND-array and re-read as [], so a second round trip lost the value + const std::vector input = + { + '{', 'U', 11, '_', 'A', 'r', 'r', 'a', 'y', 'D', 'a', 't', 'a', '_', 'Z', + 'U', 11, '_', 'A', 'r', 'r', 'a', 'y', 'T', 'y', 'p', 'e', '_', 'S', 'i', 5, 'i', 'n', 't', '1', '6', + 'U', 11, '_', 'A', 'r', 'r', 'a', 'y', 'S', 'i', 'z', 'e', '_', '[', '$', 'B', '#', '[', ']', '}' + }; + const json j1 = json::from_bjdata(input); + const json j2 = json::from_bjdata(json::to_bjdata(j1, false, false)); + CHECK(j2 == json::parse(R"({"_ArrayType_":"int16","_ArraySize_":[],"_ArrayData_":null})")); + CHECK(json::from_bjdata(json::to_bjdata(j2, false, false)) == j2); } SECTION("ndarray with _ArrayType_ \"byte\" is gated by the BJData draft version") From bc066c183810f2d8368b3675331a1e6c82165eea Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Thu, 24 Sep 2026 17:02:39 +0200 Subject: [PATCH 03/17] Make serve_header.py listen on localhost and limit its CORS header (#5564) Without a bind address in serve_header.yml, the server listened on all interfaces, so any machine on the network could fetch the header and trigger make runs in the working trees. It now listens on localhost unless configured otherwise; bind: null restores the old behavior. The header was also sent with Access-Control-Allow-Origin: *, letting any web page read it. CORS is only needed because Compiler Explorer downloads #include headers in the browser, so the header now goes only to https://godbolt.org and https://compiler-explorer.com, configurable with cors_origins. Signed-off-by: Niels Lohmann --- tools/serve_header/README.md | 4 ++++ tools/serve_header/serve_header.py | 24 +++++++++++++++++---- tools/serve_header/serve_header.yml.example | 11 ++++++++-- 3 files changed, 33 insertions(+), 6 deletions(-) diff --git a/tools/serve_header/README.md b/tools/serve_header/README.md index 0d0ed69f6..bdf2c60ec 100644 --- a/tools/serve_header/README.md +++ b/tools/serve_header/README.md @@ -60,6 +60,10 @@ int main() { `serve_header.py` will try to read a configuration file `serve_header.yml` in the top level or project root directory, and will fall back on built-in defaults if the file cannot be read. An annotated example configuration can be found in `tools/serve_header/serve_header.yml.example`. +By default, the server listens on `localhost` only, and only web pages from Compiler Explorer (`https://godbolt.org` and `https://compiler-explorer.com`) may read the header. +Set `bind` to serve other machines as well; anyone who can reach the server can then trigger `make` runs in your working trees. +Set `cors_origins` to allow other web pages. + ## Serving `json.hpp` from multiple project directory instances or working trees `serve_header.py` was designed with the goal of supporting multiple project roots or working trees at the same time. diff --git a/tools/serve_header/serve_header.py b/tools/serve_header/serve_header.py index e2da2dad0..1f29cb589 100755 --- a/tools/serve_header/serve_header.py +++ b/tools/serve_header/serve_header.py @@ -26,6 +26,10 @@ HEADER = 'json.hpp' DATETIME_FORMAT = '%Y-%m-%d %H:%M:%S' +# origins whose pages may read the served header from a browser; Compiler +# Explorer downloads #include headers client-side +DEFAULT_CORS_ORIGINS = ['https://godbolt.org', 'https://compiler-explorer.com'] + JSON_VERSION_RE = re.compile(r'\s*#\s*define\s+NLOHMANN_JSON_VERSION_MAJOR\s+') class ExitHandler(logging.StreamHandler): @@ -247,6 +251,8 @@ class WorkTrees(FileSystemEventHandler): self.observer.join() class HeaderRequestHandler(SimpleHTTPRequestHandler): # lgtm[py/missing-call-to-init] + cors_origins = DEFAULT_CORS_ORIGINS + def __init__(self, request, client_address, server): """.""" self.worktrees = server.worktrees @@ -310,8 +316,11 @@ class HeaderRequestHandler(SimpleHTTPRequestHandler): # lgtm[py/missing-call-to- # set content length super().send_header('Content-Length', length) - # CORS header - self.send_header('Access-Control-Allow-Origin', '*') + # CORS header; only for the configured origins + origin = self.headers.get('Origin') + if origin in self.cors_origins: + self.send_header('Access-Control-Allow-Origin', origin) + self.send_header('Vary', 'Origin') # prevent caching self.send_header('Cache-Control', 'no-cache, no-store, must-revalidate') self.send_header('Pragma', 'no-cache') @@ -383,8 +392,15 @@ if __name__ == '__main__': # find and monitor working trees worktrees = WorkTrees(config.get('root', '.')) - # start web server - infos = socket.getaddrinfo(config.get('bind', None), config.get('port', 8443), + # origins allowed to read the header from a browser + cors_origins = config.get('cors_origins', DEFAULT_CORS_ORIGINS) + if isinstance(cors_origins, str): + cors_origins = [cors_origins] + HeaderRequestHandler.cors_origins = cors_origins + + # start web server; only reachable from this machine unless configured + # otherwise (bind: null listens on all interfaces) + infos = socket.getaddrinfo(config.get('bind', 'localhost'), config.get('port', 8443), type=socket.SOCK_STREAM, flags=socket.AI_PASSIVE) DualStackServer.address_family = infos[0][0] HeaderRequestHandler.protocol_version = 'HTTP/1.0' diff --git a/tools/serve_header/serve_header.yml.example b/tools/serve_header/serve_header.yml.example index 42310910e..ec75e2b49 100644 --- a/tools/serve_header/serve_header.yml.example +++ b/tools/serve_header/serve_header.yml.example @@ -10,6 +10,13 @@ # cert_file: localhost.pem # key_file: localhost-key.pem -# address and port for the server to listen on -# bind: null +# address and port for the server to listen on; by default, only this machine +# can connect. Binding to a network address, or to null for all interfaces, +# lets other machines connect, and every request runs make in a working tree. +# bind: localhost # port: 8443 + +# origins whose web pages may read the header (CORS) +# cors_origins: +# - https://godbolt.org +# - https://compiler-explorer.com From 770f62dda9bb06c9772b181f7c6b452ecbe8fd9b Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Thu, 24 Sep 2026 17:02:48 +0200 Subject: [PATCH 04/17] Point to the clang-tidy check that rewrites implicit conversions (#5563) The community-maintained clang-tidy check modernize-nlohmann-json-explicit-conversions rewrites implicit conversions into explicit get() calls, which is exactly the preparation the docs ask for ahead of implicit conversions being switched off by default. Mention it on the JSON_USE_IMPLICIT_CONVERSIONS page and in the migration guide, as promised in discussion #4610. Signed-off-by: Niels Lohmann --- .../docs/api/macros/json_use_implicit_conversions.md | 8 ++++++++ docs/mkdocs/docs/integration/migration_guide.md | 6 ++++++ 2 files changed, 14 insertions(+) diff --git a/docs/mkdocs/docs/api/macros/json_use_implicit_conversions.md b/docs/mkdocs/docs/api/macros/json_use_implicit_conversions.md index 22f6d0072..e3d5fb29d 100644 --- a/docs/mkdocs/docs/api/macros/json_use_implicit_conversions.md +++ b/docs/mkdocs/docs/api/macros/json_use_implicit_conversions.md @@ -24,6 +24,14 @@ By default, implicit conversions are enabled. You can prepare existing code by already defining `JSON_USE_IMPLICIT_CONVERSIONS` to `0` and replace any implicit conversions with calls to [`get`](../basic_json/get.md). +!!! tip "Automatic migration" + + The community-maintained clang-tidy check `modernize-nlohmann-json-explicit-conversions` rewrites implicit + conversions into explicit calls to [`get`](../basic_json/get.md); for example, `#!cpp int i = j;` becomes + `#!cpp int i = j.get();`. The check is not part of clang-tidy itself, and it does not catch every case (for + example, constructing a `std::optional` from a JSON value), so review the result. See + [discussion #4610](https://github.com/nlohmann/json/discussions/4610) for how to build and use it. + !!! hint "CMake option" Implicit conversions can also be controlled with the CMake option diff --git a/docs/mkdocs/docs/integration/migration_guide.md b/docs/mkdocs/docs/integration/migration_guide.md index cb1c62d7b..8b718d969 100644 --- a/docs/mkdocs/docs/integration/migration_guide.md +++ b/docs/mkdocs/docs/integration/migration_guide.md @@ -176,6 +176,12 @@ You can prepare existing code by already defining conversions with calls to [`get`](../api/basic_json/get.md), [`get_to`](../api/basic_json/get_to.md), [`get_ref`](../api/basic_json/get_ref.md), or [`get_ptr`](../api/basic_json/get_ptr.md). +!!! tip "Automatic migration" + + The community-maintained clang-tidy check `modernize-nlohmann-json-explicit-conversions` rewrites most implicit + conversions into calls to [`get`](../api/basic_json/get.md). It is not part of clang-tidy itself; see + [discussion #4610](https://github.com/nlohmann/json/discussions/4610) for how to build and use it. + === "Deprecated" ```cpp From 36c079149a3f9009449d5cca987a404b3eedd99f Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Thu, 24 Sep 2026 17:02:57 +0200 Subject: [PATCH 05/17] Refuse to build the fuzzer drivers with NDEBUG (#5562) The fuzzer drivers check their round trips with assert(), which NDEBUG compiles away. The OSS-Fuzz build keeps assertions on today, but nothing pins that: a build change that adds NDEBUG would silently turn every round-trip check into a mere "does not crash" check. Each driver now stops the build with an #error instead, and includes itself rather than relying on json.hpp. Signed-off-by: Niels Lohmann --- tests/src/fuzzer-parse_bjdata.cpp | 6 ++++++ tests/src/fuzzer-parse_bson.cpp | 6 ++++++ tests/src/fuzzer-parse_cbor.cpp | 6 ++++++ tests/src/fuzzer-parse_json.cpp | 6 ++++++ tests/src/fuzzer-parse_msgpack.cpp | 6 ++++++ tests/src/fuzzer-parse_ubjson.cpp | 6 ++++++ 6 files changed, 36 insertions(+) diff --git a/tests/src/fuzzer-parse_bjdata.cpp b/tests/src/fuzzer-parse_bjdata.cpp index a88479933..41c51a311 100644 --- a/tests/src/fuzzer-parse_bjdata.cpp +++ b/tests/src/fuzzer-parse_bjdata.cpp @@ -46,10 +46,16 @@ The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer drivers. */ +#include #include #include #include +// the round-trip checks below are assertions; NDEBUG would compile them away +#ifdef NDEBUG + #error "the fuzzer drivers must be built without NDEBUG" +#endif + using json = nlohmann::json; // value-stable comparison for the round-trip checks below; see the note diff --git a/tests/src/fuzzer-parse_bson.cpp b/tests/src/fuzzer-parse_bson.cpp index c5f74c7cc..16f36445b 100644 --- a/tests/src/fuzzer-parse_bson.cpp +++ b/tests/src/fuzzer-parse_bson.cpp @@ -19,10 +19,16 @@ The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer drivers. */ +#include #include #include #include +// the round-trip checks below are assertions; NDEBUG would compile them away +#ifdef NDEBUG + #error "the fuzzer drivers must be built without NDEBUG" +#endif + using json = nlohmann::json; // see http://llvm.org/docs/LibFuzzer.html diff --git a/tests/src/fuzzer-parse_cbor.cpp b/tests/src/fuzzer-parse_cbor.cpp index b38e3c1e5..7d599abe2 100644 --- a/tests/src/fuzzer-parse_cbor.cpp +++ b/tests/src/fuzzer-parse_cbor.cpp @@ -19,10 +19,16 @@ The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer drivers. */ +#include #include #include #include +// the round-trip checks below are assertions; NDEBUG would compile them away +#ifdef NDEBUG + #error "the fuzzer drivers must be built without NDEBUG" +#endif + using json = nlohmann::json; // see http://llvm.org/docs/LibFuzzer.html diff --git a/tests/src/fuzzer-parse_json.cpp b/tests/src/fuzzer-parse_json.cpp index 59a278c7b..217d9e0fd 100644 --- a/tests/src/fuzzer-parse_json.cpp +++ b/tests/src/fuzzer-parse_json.cpp @@ -20,10 +20,16 @@ The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer drivers. */ +#include #include #include #include +// the round-trip checks below are assertions; NDEBUG would compile them away +#ifdef NDEBUG + #error "the fuzzer drivers must be built without NDEBUG" +#endif + using json = nlohmann::json; // see http://llvm.org/docs/LibFuzzer.html diff --git a/tests/src/fuzzer-parse_msgpack.cpp b/tests/src/fuzzer-parse_msgpack.cpp index 0b4ab0af7..df961b8d7 100644 --- a/tests/src/fuzzer-parse_msgpack.cpp +++ b/tests/src/fuzzer-parse_msgpack.cpp @@ -19,10 +19,16 @@ The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer drivers. */ +#include #include #include #include +// the round-trip checks below are assertions; NDEBUG would compile them away +#ifdef NDEBUG + #error "the fuzzer drivers must be built without NDEBUG" +#endif + using json = nlohmann::json; // see http://llvm.org/docs/LibFuzzer.html diff --git a/tests/src/fuzzer-parse_ubjson.cpp b/tests/src/fuzzer-parse_ubjson.cpp index 463656c71..20c20eda7 100644 --- a/tests/src/fuzzer-parse_ubjson.cpp +++ b/tests/src/fuzzer-parse_ubjson.cpp @@ -25,10 +25,16 @@ The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer drivers. */ +#include #include #include #include +// the round-trip checks below are assertions; NDEBUG would compile them away +#ifdef NDEBUG + #error "the fuzzer drivers must be built without NDEBUG" +#endif + using json = nlohmann::json; // see http://llvm.org/docs/LibFuzzer.html From 5f659c881a7f08b1664ff7ef600644fd8619034a Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Thu, 24 Sep 2026 17:03:48 +0200 Subject: [PATCH 06/17] Let the labeler assign "aspect: binary formats", "python" and more "CI" (#5557) - "aspect: binary formats" for changes to the binary reader or writer, their tests, fuzzers and docs, or with a binary format in the title; - "python" for Python sources and pip requirements files, matching the label Dependabot sets on its pip updates, so it is never removed there; - "CI" also for changes to the Dependabot and labeler configurations. Signed-off-by: Niels Lohmann --- .github/labeler.yml | 21 +++++++++++++++++++++ 1 file changed, 21 insertions(+) diff --git a/.github/labeler.yml b/.github/labeler.yml index 828660daf..b4c176960 100644 --- a/.github/labeler.yml +++ b/.github/labeler.yml @@ -29,6 +29,27 @@ labels: files: - ".github/external_ci/.*" +- label: "CI" + files: + - ".github/(dependabot|labeler)\\.yml" + +- label: "aspect: binary formats" + files: + - "include/nlohmann/detail/input/binary_reader\\.hpp" + - "include/nlohmann/detail/output/binary_writer\\.hpp" + - "tests/src/unit-(bson|cbor|msgpack|ubjson|bjdata|binary_formats)" + - "tests/src/fuzzer-parse_(bson|cbor|msgpack|ubjson|bjdata)" + - "docs/mkdocs/docs/features/binary_formats/" + - "docs/mkdocs/docs/(api/basic_json|examples)/(to|from)_(bson|cbor|msgpack|ubjson|bjdata)" + +- label: "aspect: binary formats" + title: "(?i)(bson|cbor|msgpack|messagepack|ubjson|bjdata|binary format)" + +- label: "python" + files: + - "\\.py$" + - "requirements[^/]*\\.txt$" + - label: "S" size-below: 10 - label: "M" From 8699de30642500fb8c940aee9c9812bb3f5af857 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Thu, 24 Sep 2026 17:04:44 +0200 Subject: [PATCH 07/17] Stop allocating the BJData excluded-marker list per container (#5555) write_ubjson() built a std::vector of the eight markers BJData forbids as the type of an optimized container - one heap allocation plus a linear search for every array and object it wrote with use_type, even for plain UBJSON output, where the list isn't consulted. The list was also spelled out twice. A constexpr helper, is_bjdata_excluded_type_marker(), replaces both. Signed-off-by: Niels Lohmann --- .../nlohmann/detail/output/binary_writer.hpp | 23 ++++++++++++++----- single_include/nlohmann/json.hpp | 23 ++++++++++++++----- 2 files changed, 34 insertions(+), 12 deletions(-) diff --git a/include/nlohmann/detail/output/binary_writer.hpp b/include/nlohmann/detail/output/binary_writer.hpp index 495f872c0..eaa8d63b6 100644 --- a/include/nlohmann/detail/output/binary_writer.hpp +++ b/include/nlohmann/detail/output/binary_writer.hpp @@ -881,8 +881,6 @@ class binary_writer return ubjson_prefix(v, use_bjdata) == first_prefix; }); - std::vector bjdx = {'[', '{', 'S', 'H', 'T', 'F', 'N', 'Z'}; // excluded markers in bjdata optimized type - // an optimized array of a valueless type carries no payload, so a // reader has nothing but the declared count to bound the allocation // by and refuses an excessive one. Write the unoptimized form for @@ -893,7 +891,7 @@ class binary_writer && j.m_data.m_value.array->size() > detail::max_valueless_container_size; if (same_prefix && !excessive_valueless - && !(use_bjdata && std::find(bjdx.begin(), bjdx.end(), first_prefix) != bjdx.end())) + && !(use_bjdata && is_bjdata_excluded_type_marker(first_prefix))) { prefix_required = false; oa.write_character(to_char_type('$')); @@ -997,9 +995,7 @@ class binary_writer return ubjson_prefix(v, use_bjdata) == first_prefix; }); - std::vector bjdx = {'[', '{', 'S', 'H', 'T', 'F', 'N', 'Z'}; // excluded markers in bjdata optimized type - - if (same_prefix && !(use_bjdata && std::find(bjdx.begin(), bjdx.end(), first_prefix) != bjdx.end())) + if (same_prefix && !(use_bjdata && is_bjdata_excluded_type_marker(first_prefix))) { prefix_required = false; oa.write_character(to_char_type('$')); @@ -1726,6 +1722,21 @@ class binary_writer } } + /*! + @brief whether BJData forbids @a marker as the type of an optimized array + or object + + Containers, strings, high-precision numbers, booleans and null cannot be + declared as the single type of an optimized container in BJData; such a + container is written unoptimized. The reader rejects them with the same + list (binary_reader::bjd_optimized_type_markers). + */ + static constexpr bool is_bjdata_excluded_type_marker(const CharType marker) noexcept + { + return marker == '[' || marker == '{' || marker == 'S' || marker == 'H' + || marker == 'T' || marker == 'F' || marker == 'N' || marker == 'Z'; + } + static constexpr CharType get_ubjson_float_prefix(float /*unused*/) { return 'd'; // float 32 diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 1d5e21927..aa2916aa5 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -19745,8 +19745,6 @@ class binary_writer return ubjson_prefix(v, use_bjdata) == first_prefix; }); - std::vector bjdx = {'[', '{', 'S', 'H', 'T', 'F', 'N', 'Z'}; // excluded markers in bjdata optimized type - // an optimized array of a valueless type carries no payload, so a // reader has nothing but the declared count to bound the allocation // by and refuses an excessive one. Write the unoptimized form for @@ -19757,7 +19755,7 @@ class binary_writer && j.m_data.m_value.array->size() > detail::max_valueless_container_size; if (same_prefix && !excessive_valueless - && !(use_bjdata && std::find(bjdx.begin(), bjdx.end(), first_prefix) != bjdx.end())) + && !(use_bjdata && is_bjdata_excluded_type_marker(first_prefix))) { prefix_required = false; oa.write_character(to_char_type('$')); @@ -19861,9 +19859,7 @@ class binary_writer return ubjson_prefix(v, use_bjdata) == first_prefix; }); - std::vector bjdx = {'[', '{', 'S', 'H', 'T', 'F', 'N', 'Z'}; // excluded markers in bjdata optimized type - - if (same_prefix && !(use_bjdata && std::find(bjdx.begin(), bjdx.end(), first_prefix) != bjdx.end())) + if (same_prefix && !(use_bjdata && is_bjdata_excluded_type_marker(first_prefix))) { prefix_required = false; oa.write_character(to_char_type('$')); @@ -20590,6 +20586,21 @@ class binary_writer } } + /*! + @brief whether BJData forbids @a marker as the type of an optimized array + or object + + Containers, strings, high-precision numbers, booleans and null cannot be + declared as the single type of an optimized container in BJData; such a + container is written unoptimized. The reader rejects them with the same + list (binary_reader::bjd_optimized_type_markers). + */ + static constexpr bool is_bjdata_excluded_type_marker(const CharType marker) noexcept + { + return marker == '[' || marker == '{' || marker == 'S' || marker == 'H' + || marker == 'T' || marker == 'F' || marker == 'N' || marker == 'Z'; + } + static constexpr CharType get_ubjson_float_prefix(float /*unused*/) { return 'd'; // float 32 From 2e91641de27335ff48bca48c2b2281f2688b1e99 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Thu, 24 Sep 2026 17:05:37 +0200 Subject: [PATCH 08/17] Test JSON_BRACE_INIT_COPY_SEMANTICS for real, and fix one-element tuples under it (#5544) * Test JSON_BRACE_INIT_COPY_SEMANTICS for real, and fix one-element tuples under it The opt-in JSON_BRACE_INIT_COPY_SEMANTICS was never exercised by CI: - Its only test, in unit-regression3.cpp, was guarded by `#if defined(JSON_BRACE_INIT_COPY_SEMANTICS)` after the #include. The header #undefs the macro unconditionally in macro_unscope.hpp, so the guard was always false and the test compiled to nothing, whatever -D flag was passed. - The ci_test_brace_init_copy_semantics target that passes the flag was not named by any workflow. Move the test into its own translation unit that defines the macro before including the header, as unit-diagnostics.cpp does for JSON_DIAGNOSTICS. It now runs in every CI job and for every standard. Remove the unused target: it ran the whole suite with the macro, and that suite deliberately relies on default brace-init semantics in about 90 places (e.g. `json({1})` meaning `[1]`), so it could never pass. Running the whole suite with the macro did find one library bug: to_json for std::tuple builds `j = { std::get(t)... }`, so with copy semantics a one-element tuple became its element. `json(std::tuple{5})` was `5` instead of `[5]`, and `get>()` threw type_error.302 on the result. Under the macro, a one-element tuple now builds exactly what the default deduction builds. Without the macro nothing changes. The new tests also pin that the library's other conversions produce the same values with and without the macro. The macro page now says that the macro affects every single-element list (`json j = {1}` is `1`), and that all translation units must agree on it, since it has no ABI tag. Signed-off-by: Niels Lohmann * Make JSON_BRACE_INIT_COPY_SEMANTICS part of the ABI tag The macro changes the body of the initializer-list constructor and adds a to_json_tuple_impl overload, both with the same mangled names in either mode, so mixing translation units silently picked one definition. Encode it in the inline namespace as `_bics`, as JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON does with `_ldvcmp`. The macro is new in the unreleased 3.13.0, so no existing namespace name changes. - Move the macro's default into abi_macros.hpp so json_fwd.hpp computes the same namespace, and keep it defined under JSON_TEST_KEEP_MACROS. - Check the tag in the ABI config tests and in the unit test. - List `_bics` (and the missing `_dp`) in the namespace docs and in the natvis generator; regenerate nlohmann_json.natvis. - Replace the "define it consistently" warning with an ABI note. Suggested by @gregmarr in the review of #5544. Signed-off-by: Niels Lohmann * Fix the cppcheck, clang-tidy and legacy-comparison CI failures - to_json_tuple_impl() moved the element in both branches of a ternary; only one runs, but cppcheck reported accessMoved. Use if/else. - The ABI tag test looked for "json_abi_bics", which misses when another tag comes first, as in json_abi_ldvcmp_bics; look for "_bics". - readability-qualified-auto in the items() test. Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- cmake/ci.cmake | 15 - .../macros/json_brace_init_copy_semantics.md | 22 + docs/mkdocs/docs/features/namespace.md | 2 + include/nlohmann/detail/abi_macros.hpp | 19 +- .../nlohmann/detail/conversions/to_json.hpp | 24 + include/nlohmann/detail/macro_scope.hpp | 4 - include/nlohmann/detail/macro_unscope.hpp | 2 +- nlohmann_json.natvis | 720 ++++++++++++++++++ single_include/nlohmann/json.hpp | 49 +- single_include/nlohmann/json_fwd.hpp | 19 +- tests/abi/config/default.cpp | 4 + tests/abi/config/noversion.cpp | 4 + tests/src/unit-brace-init-copy-semantics.cpp | 167 ++++ tests/src/unit-regression3.cpp | 27 - tools/generate_natvis/generate_natvis.py | 2 +- 15 files changed, 1015 insertions(+), 65 deletions(-) create mode 100644 tests/src/unit-brace-init-copy-semantics.cpp diff --git a/cmake/ci.cmake b/cmake/ci.cmake index f854138b6..a99788633 100644 --- a/cmake/ci.cmake +++ b/cmake/ci.cmake @@ -230,21 +230,6 @@ add_custom_target(ci_test_simdutf COMMENT "Compile and test with simdutf UTF-8 validation enabled" ) -############################################################################### -# Enable brace-init copy semantics. -############################################################################### - -add_custom_target(ci_test_brace_init_copy_semantics - COMMAND ${CMAKE_COMMAND} - -DCMAKE_BUILD_TYPE=Debug -GNinja - -DJSON_BuildTests=ON -DJSON_FastTests=ON - -DCMAKE_CXX_FLAGS=-DJSON_BRACE_INIT_COPY_SEMANTICS=1 - -S${PROJECT_SOURCE_DIR} -B${PROJECT_BINARY_DIR}/build_brace_init_copy_semantics - COMMAND ${CMAKE_COMMAND} --build ${PROJECT_BINARY_DIR}/build_brace_init_copy_semantics - COMMAND cd ${PROJECT_BINARY_DIR}/build_brace_init_copy_semantics && ${CMAKE_CTEST_COMMAND} --parallel ${N} --output-on-failure - COMMENT "Compile and test with brace-init copy semantics enabled" -) - ############################################################################### # Enable strict NUL-byte handling. ############################################################################### diff --git a/docs/mkdocs/docs/api/macros/json_brace_init_copy_semantics.md b/docs/mkdocs/docs/api/macros/json_brace_init_copy_semantics.md index 970c20537..2301a0486 100644 --- a/docs/mkdocs/docs/api/macros/json_brace_init_copy_semantics.md +++ b/docs/mkdocs/docs/api/macros/json_brace_init_copy_semantics.md @@ -38,6 +38,28 @@ The default value is `0` (disabled — existing behavior is preserved). This macro must be defined **before** including ``. Defining it after the include has no effect. +!!! warning "Applies to every single-element list" + + The macro does not only affect a single JSON value in braces. **Any** single-element braced list is treated as its + element, so it no longer creates a one-element array: + + ```cpp + json j1 = {1}; // 1, not [1] + json j2 = {"text"}; // "text", not ["text"] + json j3 = {{1, 2}}; // [1,2], not [[1,2]] + ``` + + Code that relies on these producing arrays must use `json::array()` instead (see below). Lists with more than one + element, and a single `[string, value]` pair such as `{{"key", "value"}}`, which still creates an object, are not + affected. The library's own conversions are not affected either: for example, `std::tuple{5}` still becomes + `[5]`. + +!!! note "ABI compatibility" + + The value of this macro is encoded in the [namespace](../../features/namespace.md) (tag `_bics`), resulting in + distinct symbol names. Translation units compiled with and without it can therefore be linked into the same program + without One Definition Rule (ODR) violations, but they cannot exchange instances of library types. + !!! tip "Workaround without the macro" To explicitly create a single-element array without enabling this macro, use `json::array()`: diff --git a/docs/mkdocs/docs/features/namespace.md b/docs/mkdocs/docs/features/namespace.md index c4efe772a..09e53f3a2 100644 --- a/docs/mkdocs/docs/features/namespace.md +++ b/docs/mkdocs/docs/features/namespace.md @@ -16,6 +16,8 @@ The complete default namespace name is derived as follows: - [`JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON`](../api/macros/json_use_legacy_discarded_value_comparison.md) defined non-zero appends `_ldvcmp`. - [`JSON_DIAGNOSTIC_POSITIONS`](../api/macros/json_diagnostic_positions.md) defined non-zero appends `_dp`. + - [`JSON_BRACE_INIT_COPY_SEMANTICS`](../api/macros/json_brace_init_copy_semantics.md) defined non-zero appends + `_bics`. - The inline namespace ends with the suffix `_v` followed by the 3 components of the version number separated by underscores. To omit the version component, see [Disabling the version component](#disabling-the-version-component) below. diff --git a/include/nlohmann/detail/abi_macros.hpp b/include/nlohmann/detail/abi_macros.hpp index 3e07a6a98..cca04e8ec 100644 --- a/include/nlohmann/detail/abi_macros.hpp +++ b/include/nlohmann/detail/abi_macros.hpp @@ -34,6 +34,10 @@ #define JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON 0 #endif +#ifndef JSON_BRACE_INIT_COPY_SEMANTICS + #define JSON_BRACE_INIT_COPY_SEMANTICS 0 +#endif + #if JSON_DIAGNOSTICS #define NLOHMANN_JSON_ABI_TAG_DIAGNOSTICS _diag #else @@ -52,20 +56,27 @@ #define NLOHMANN_JSON_ABI_TAG_LEGACY_DISCARDED_VALUE_COMPARISON #endif +#if JSON_BRACE_INIT_COPY_SEMANTICS + #define NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS _bics +#else + #define NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS +#endif + #ifndef NLOHMANN_JSON_NAMESPACE_NO_VERSION #define NLOHMANN_JSON_NAMESPACE_NO_VERSION 0 #endif // Construct the namespace ABI tags component -#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c) json_abi ## a ## b ## c -#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c) \ - NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c) +#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d) json_abi ## a ## b ## c ## d +#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d) \ + NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d) #define NLOHMANN_JSON_ABI_TAGS \ NLOHMANN_JSON_ABI_TAGS_CONCAT( \ NLOHMANN_JSON_ABI_TAG_DIAGNOSTICS, \ NLOHMANN_JSON_ABI_TAG_LEGACY_DISCARDED_VALUE_COMPARISON, \ - NLOHMANN_JSON_ABI_TAG_DIAGNOSTIC_POSITIONS) + NLOHMANN_JSON_ABI_TAG_DIAGNOSTIC_POSITIONS, \ + NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS) // Construct the namespace version component #define NLOHMANN_JSON_NAMESPACE_VERSION_CONCAT_EX(major, minor, patch) \ diff --git a/include/nlohmann/detail/conversions/to_json.hpp b/include/nlohmann/detail/conversions/to_json.hpp index 5f8644700..491bb9873 100644 --- a/include/nlohmann/detail/conversions/to_json.hpp +++ b/include/nlohmann/detail/conversions/to_json.hpp @@ -471,6 +471,30 @@ inline void to_json_tuple_impl(BasicJsonType& j, const Tuple& t, index_sequence< j = { std::get(t)... }; } +#if JSON_BRACE_INIT_COPY_SEMANTICS +// JSON_BRACE_INIT_COPY_SEMANTICS makes a one-element braced list copy its +// element instead of wrapping it, which would serialize std::tuple{5} as 5 +// rather than [5]. Build what the default deduction builds instead: an object +// if the element is a [string, value] pair, a one-element array otherwise. +template +inline void to_json_tuple_impl(BasicJsonType& j, const Tuple& t, index_sequence<0> /*unused*/) +{ + BasicJsonType element(std::get<0>(t)); + // same test as the initializer-list constructor, including the cast that + // keeps a string type constructible from 0 from selecting operator[](key) + const bool is_member = element.is_array() && element.size() == 2 + && element[static_cast(0)].is_string(); + if (is_member) + { + j = BasicJsonType::object({std::move(element)}); + } + else + { + j = BasicJsonType::array({std::move(element)}); + } +} +#endif + template inline void to_json_tuple_impl(BasicJsonType& j, const Tuple& /*unused*/, index_sequence<> /*unused*/) { diff --git a/include/nlohmann/detail/macro_scope.hpp b/include/nlohmann/detail/macro_scope.hpp index def9da6f8..96fa165f5 100644 --- a/include/nlohmann/detail/macro_scope.hpp +++ b/include/nlohmann/detail/macro_scope.hpp @@ -813,10 +813,6 @@ void templated_json_throw(ExceptionType exception) #define JSON_USE_GLOBAL_UDLS 1 #endif -#ifndef JSON_BRACE_INIT_COPY_SEMANTICS - #define JSON_BRACE_INIT_COPY_SEMANTICS 0 -#endif - #ifndef JSON_STRICT_NUL_HANDLING #define JSON_STRICT_NUL_HANDLING 0 #endif diff --git a/include/nlohmann/detail/macro_unscope.hpp b/include/nlohmann/detail/macro_unscope.hpp index c692ea68e..afcbfc38b 100644 --- a/include/nlohmann/detail/macro_unscope.hpp +++ b/include/nlohmann/detail/macro_unscope.hpp @@ -26,7 +26,6 @@ #undef JSON_NO_UNIQUE_ADDRESS #undef JSON_DISABLE_ENUM_SERIALIZATION #undef JSON_USE_GLOBAL_UDLS -#undef JSON_BRACE_INIT_COPY_SEMANTICS #undef JSON_STRICT_NUL_HANDLING #ifndef JSON_TEST_KEEP_MACROS @@ -45,6 +44,7 @@ #undef JSON_HAS_STD_FORMAT #undef JSON_HAS_STATIC_RTTI #undef JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON + #undef JSON_BRACE_INIT_COPY_SEMANTICS #endif #include diff --git a/nlohmann_json.natvis b/nlohmann_json.natvis index 09a46d67d..2eccbe17c 100644 --- a/nlohmann_json.natvis +++ b/nlohmann_json.natvis @@ -215,6 +215,126 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -275,4 +395,604 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index aa2916aa5..c11fa27e5 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -91,6 +91,10 @@ #define JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON 0 #endif +#ifndef JSON_BRACE_INIT_COPY_SEMANTICS + #define JSON_BRACE_INIT_COPY_SEMANTICS 0 +#endif + #if JSON_DIAGNOSTICS #define NLOHMANN_JSON_ABI_TAG_DIAGNOSTICS _diag #else @@ -109,20 +113,27 @@ #define NLOHMANN_JSON_ABI_TAG_LEGACY_DISCARDED_VALUE_COMPARISON #endif +#if JSON_BRACE_INIT_COPY_SEMANTICS + #define NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS _bics +#else + #define NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS +#endif + #ifndef NLOHMANN_JSON_NAMESPACE_NO_VERSION #define NLOHMANN_JSON_NAMESPACE_NO_VERSION 0 #endif // Construct the namespace ABI tags component -#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c) json_abi ## a ## b ## c -#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c) \ - NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c) +#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d) json_abi ## a ## b ## c ## d +#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d) \ + NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d) #define NLOHMANN_JSON_ABI_TAGS \ NLOHMANN_JSON_ABI_TAGS_CONCAT( \ NLOHMANN_JSON_ABI_TAG_DIAGNOSTICS, \ NLOHMANN_JSON_ABI_TAG_LEGACY_DISCARDED_VALUE_COMPARISON, \ - NLOHMANN_JSON_ABI_TAG_DIAGNOSTIC_POSITIONS) + NLOHMANN_JSON_ABI_TAG_DIAGNOSTIC_POSITIONS, \ + NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS) // Construct the namespace version component #define NLOHMANN_JSON_NAMESPACE_VERSION_CONCAT_EX(major, minor, patch) \ @@ -3191,10 +3202,6 @@ void templated_json_throw(ExceptionType exception) #define JSON_USE_GLOBAL_UDLS 1 #endif -#ifndef JSON_BRACE_INIT_COPY_SEMANTICS - #define JSON_BRACE_INIT_COPY_SEMANTICS 0 -#endif - #ifndef JSON_STRICT_NUL_HANDLING #define JSON_STRICT_NUL_HANDLING 0 #endif @@ -6767,6 +6774,30 @@ inline void to_json_tuple_impl(BasicJsonType& j, const Tuple& t, index_sequence< j = { std::get(t)... }; } +#if JSON_BRACE_INIT_COPY_SEMANTICS +// JSON_BRACE_INIT_COPY_SEMANTICS makes a one-element braced list copy its +// element instead of wrapping it, which would serialize std::tuple{5} as 5 +// rather than [5]. Build what the default deduction builds instead: an object +// if the element is a [string, value] pair, a one-element array otherwise. +template +inline void to_json_tuple_impl(BasicJsonType& j, const Tuple& t, index_sequence<0> /*unused*/) +{ + BasicJsonType element(std::get<0>(t)); + // same test as the initializer-list constructor, including the cast that + // keeps a string type constructible from 0 from selecting operator[](key) + const bool is_member = element.is_array() && element.size() == 2 + && element[static_cast(0)].is_string(); + if (is_member) + { + j = BasicJsonType::object({std::move(element)}); + } + else + { + j = BasicJsonType::array({std::move(element)}); + } +} +#endif + template inline void to_json_tuple_impl(BasicJsonType& j, const Tuple& /*unused*/, index_sequence<> /*unused*/) { @@ -30395,7 +30426,6 @@ struct formatter // NOLINT(cert-dcl58-c #undef JSON_NO_UNIQUE_ADDRESS #undef JSON_DISABLE_ENUM_SERIALIZATION #undef JSON_USE_GLOBAL_UDLS -#undef JSON_BRACE_INIT_COPY_SEMANTICS #undef JSON_STRICT_NUL_HANDLING #ifndef JSON_TEST_KEEP_MACROS @@ -30414,6 +30444,7 @@ struct formatter // NOLINT(cert-dcl58-c #undef JSON_HAS_STD_FORMAT #undef JSON_HAS_STATIC_RTTI #undef JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON + #undef JSON_BRACE_INIT_COPY_SEMANTICS #endif // #include diff --git a/single_include/nlohmann/json_fwd.hpp b/single_include/nlohmann/json_fwd.hpp index 525e65b64..281c05efa 100644 --- a/single_include/nlohmann/json_fwd.hpp +++ b/single_include/nlohmann/json_fwd.hpp @@ -52,6 +52,10 @@ #define JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON 0 #endif +#ifndef JSON_BRACE_INIT_COPY_SEMANTICS + #define JSON_BRACE_INIT_COPY_SEMANTICS 0 +#endif + #if JSON_DIAGNOSTICS #define NLOHMANN_JSON_ABI_TAG_DIAGNOSTICS _diag #else @@ -70,20 +74,27 @@ #define NLOHMANN_JSON_ABI_TAG_LEGACY_DISCARDED_VALUE_COMPARISON #endif +#if JSON_BRACE_INIT_COPY_SEMANTICS + #define NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS _bics +#else + #define NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS +#endif + #ifndef NLOHMANN_JSON_NAMESPACE_NO_VERSION #define NLOHMANN_JSON_NAMESPACE_NO_VERSION 0 #endif // Construct the namespace ABI tags component -#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c) json_abi ## a ## b ## c -#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c) \ - NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c) +#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d) json_abi ## a ## b ## c ## d +#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d) \ + NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d) #define NLOHMANN_JSON_ABI_TAGS \ NLOHMANN_JSON_ABI_TAGS_CONCAT( \ NLOHMANN_JSON_ABI_TAG_DIAGNOSTICS, \ NLOHMANN_JSON_ABI_TAG_LEGACY_DISCARDED_VALUE_COMPARISON, \ - NLOHMANN_JSON_ABI_TAG_DIAGNOSTIC_POSITIONS) + NLOHMANN_JSON_ABI_TAG_DIAGNOSTIC_POSITIONS, \ + NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS) // Construct the namespace version component #define NLOHMANN_JSON_NAMESPACE_VERSION_CONCAT_EX(major, minor, patch) \ diff --git a/tests/abi/config/default.cpp b/tests/abi/config/default.cpp index f3ee23110..d0b4ba54b 100644 --- a/tests/abi/config/default.cpp +++ b/tests/abi/config/default.cpp @@ -32,6 +32,10 @@ TEST_CASE("default namespace") expected += "_dp"; #endif +#if JSON_BRACE_INIT_COPY_SEMANTICS + expected += "_bics"; +#endif + expected += "_v" STRINGIZE(NLOHMANN_JSON_VERSION_MAJOR); expected += "_" STRINGIZE(NLOHMANN_JSON_VERSION_MINOR); expected += "_" STRINGIZE(NLOHMANN_JSON_VERSION_PATCH) "::basic_json"; diff --git a/tests/abi/config/noversion.cpp b/tests/abi/config/noversion.cpp index cbdcb149b..789107181 100644 --- a/tests/abi/config/noversion.cpp +++ b/tests/abi/config/noversion.cpp @@ -33,6 +33,10 @@ TEST_CASE("default namespace without version component") expected += "_dp"; #endif +#if JSON_BRACE_INIT_COPY_SEMANTICS + expected += "_bics"; +#endif + expected += "::basic_json"; // fallback for Clang diff --git a/tests/src/unit-brace-init-copy-semantics.cpp b/tests/src/unit-brace-init-copy-semantics.cpp new file mode 100644 index 000000000..1ee0c6607 --- /dev/null +++ b/tests/src/unit-brace-init-copy-semantics.cpp @@ -0,0 +1,167 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#include "doctest_compatibility.h" + +// This file tests the opt-in JSON_BRACE_INIT_COPY_SEMANTICS, so it defines the +// macro itself rather than relying on a -D flag, and runs in every build. +#ifdef JSON_BRACE_INIT_COPY_SEMANTICS + #undef JSON_BRACE_INIT_COPY_SEMANTICS +#endif + +#define JSON_BRACE_INIT_COPY_SEMANTICS 1 + +#include +using nlohmann::json; + +#include +#include +#include +#include +#include +#include +#include + +#define STRINGIZE_EX(x) #x +#define STRINGIZE(x) STRINGIZE_EX(x) + +TEST_CASE("JSON_BRACE_INIT_COPY_SEMANTICS") +{ + SECTION("the macro is part of the ABI tag") + { + const std::string ns = STRINGIZE(NLOHMANN_JSON_NAMESPACE); + // other tags may come before it, e.g. json_abi_ldvcmp_bics + CHECK(ns.find("_bics") != std::string::npos); + } + + SECTION("single-element brace initialization copies the element (#5074)") + { + json const j_obj = {{"key", "value"}, {"num", 42}}; + json const j_arr = {1, 2, 3}; + + // object: brace init copies instead of wrapping + json const j1{j_obj}; + CHECK(j1.is_object()); + CHECK(j1 == j_obj); + + // array: brace init copies instead of wrapping + json const j2{j_arr}; + CHECK(j2.is_array()); + CHECK(j2.size() == 3); + CHECK(j2 == j_arr); + + // this applies to any single element, not only to JSON values + json const j3{true}; + CHECK(j3.is_boolean()); + + json const j4{42}; + CHECK(j4.is_number_integer()); + + json const j5 = {1}; + CHECK(j5 == 1); + + json const j6 = {"text"}; + CHECK(j6 == "text"); + + json const j7 = {{1, 2}}; + CHECK(j7 == json::array({1, 2})); + } + + SECTION("what the macro does not change") + { + // lists with more than one element are unaffected + json const j1 = {1, 2}; + CHECK(j1.is_array()); + CHECK(j1.size() == 2); + + // a single [string, value] pair still describes an object + json const j2 = {{"key", "value"}}; + CHECK(j2.is_object()); + CHECK(j2["key"] == "value"); + + // json::array() always creates an array + json const j3 = json::array({1}); + CHECK(j3.is_array()); + CHECK(j3.size() == 1); + CHECK(j3[0] == 1); + + json const j_obj = {{"key", "value"}}; + json const j4 = json::array({j_obj}); + CHECK(j4.is_array()); + CHECK(j4.size() == 1); + CHECK(j4[0] == j_obj); + } + + SECTION("conversions build the same values as without the macro") + { + SECTION("one-element std::tuple") + { + json const j1 = std::tuple {5}; + CHECK(j1.dump() == "[5]"); + CHECK(std::get<0>(j1.get>()) == 5); + + json const j2 = std::tuple {"text"}; + CHECK(j2.dump() == "[\"text\"]"); + CHECK(std::get<0>(j2.get>()) == "text"); + + json const j3 = std::tuple {json::array({1, 2})}; + CHECK(j3.dump() == "[[1,2]]"); + + // as without the macro, a [string, value] pair becomes an object + // member (see the known limitation documented for std::pair) + json const j4 = std::tuple> {{"a", 1}}; + CHECK(j4.dump() == "{\"a\":1}"); + } + + SECTION("tuples with more elements") + { + json const j1 = std::tuple {1, "a"}; + CHECK(j1.dump() == "[1,\"a\"]"); + + json const j2 = std::tuple<> {}; + CHECK(j2.dump() == "[]"); + } + + SECTION("one-element containers") + { + json const j1 = std::vector {1}; + CHECK(j1.dump() == "[1]"); + CHECK(j1.get>() == std::vector {1}); + + std::array const arr = {{1}}; + json const j2 = arr; + CHECK(j2.dump() == "[1]"); + + json const j3 = std::list {"a"}; + CHECK(j3.dump() == "[\"a\"]"); + + json const j4 = std::map {{"a", 1}}; + CHECK(j4.dump() == "{\"a\":1}"); + + json const j5 = std::map {{1, 2}}; + CHECK(j5.dump() == "[[1,2]]"); + } + + SECTION("std::pair") + { + json const j = std::pair {1, 2}; + CHECK(j.dump() == "[1,2]"); + CHECK((j.get>() == std::pair {1, 2})); + } + + SECTION("items()") + { + json j_obj = {{"key", 1}}; + for (const auto& el : j_obj.items()) + { + json const j = el; + CHECK(j.dump() == "{\"key\":1}"); + } + } + } +} diff --git a/tests/src/unit-regression3.cpp b/tests/src/unit-regression3.cpp index a5be9ec4b..882b3866b 100644 --- a/tests/src/unit-regression3.cpp +++ b/tests/src/unit-regression3.cpp @@ -658,33 +658,6 @@ TEST_CASE("regression test #5074 - portable workaround for single-element brace CHECK(j[0] == j_obj); } -#if defined(JSON_BRACE_INIT_COPY_SEMANTICS) && (JSON_BRACE_INIT_COPY_SEMANTICS == 1) -TEST_CASE("regression test #5074 - single-element brace init with JSON_BRACE_INIT_COPY_SEMANTICS") -{ - // with JSON_BRACE_INIT_COPY_SEMANTICS: single-element brace init copies/moves - json const j_obj = {{"key", "value"}, {"num", 42}}; - json const j_arr = {1, 2, 3}; - - // object: brace init copies instead of wrapping - json const j1{j_obj}; - CHECK(j1.is_object()); - CHECK(j1 == j_obj); - - // array: brace init copies instead of wrapping - json const j2{j_arr}; - CHECK(j2.is_array()); - CHECK(j2.size() == 3); - CHECK(j2 == j_arr); - - // primitives still work as initializer lists - json const j3{true}; - CHECK(j3.is_boolean()); - - json const j4{42}; - CHECK(j4.is_number_integer()); -} -#endif - struct Example_5122 { float b = 2; diff --git a/tools/generate_natvis/generate_natvis.py b/tools/generate_natvis/generate_natvis.py index 9266050c5..968690abe 100755 --- a/tools/generate_natvis/generate_natvis.py +++ b/tools/generate_natvis/generate_natvis.py @@ -20,7 +20,7 @@ if __name__ == '__main__': namespaces = ['nlohmann'] abi_prefix = 'json_abi' - abi_tags = ['_diag', '_ldvcmp'] + abi_tags = ['_diag', '_ldvcmp', '_dp', '_bics'] version = '_v' + args.version.replace('.', '_') inline_namespaces = [] From 305ca7dadd17bbbf215822e8e8e9c792c182cdaa Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Thu, 24 Sep 2026 17:07:21 +0200 Subject: [PATCH 09/17] Add missing headers to BUILD.bazel and check it in CI (#5554) * Add missing headers to BUILD.bazel and make its generator reproduce it The "json" cc_library did not list three headers that the library includes: - detail/meta/logic.hpp (added in #5016, included by from_json.hpp) - detail/input/number_parse.hpp (added in #5283, included by lexer.hpp) - detail/input/string_scan.hpp (added in #5283, included by lexer.hpp and serializer.hpp) Bazel's sandbox only exposes declared headers, so any target depending on @nlohmann_json//:json and including failed with "'nlohmann/detail/meta/logic.hpp' file not found". The file could not simply be regenerated, because the generator behind "make BUILD.bazel" was stale: it wrote only the "json" cc_library and dropped the load() statements, the license block, and the "singleheader-json" target that were added by hand in #4584. The generator now emits the complete file, so its output differs from the previous BUILD.bazel only by the three headers. It also resolves the glob against the project root instead of the working directory and sorts the list explicitly. "make BUILD.bazel" is now phony: in a fresh checkout, BUILD.bazel is not older than the headers, so make considered it up to date, and a removed header would never trigger a rebuild. "make check-amalgamation" also checks that BUILD.bazel is up to date. Signed-off-by: Niels Lohmann * Check in CI that BUILD.bazel is up to date The "Check amalgamation" workflow now also regenerates BUILD.bazel, so a pull request that adds, renames, or removes a header without updating the Bazel header list fails, and the attached amalgamation.patch contains the fix. The failure comment and the contribution guidelines mention the new check, and the comment now links to the existing "Amalgamate the source code" section instead of the "Files to change" anchor that was removed in #4560. Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- .github/CONTRIBUTING.md | 9 ++++ .github/workflows/check_amalgamation.yml | 7 ++- .../workflows/comment_check_amalgamation.yml | 4 +- BUILD.bazel | 3 ++ FILES.md | 6 ++- Makefile | 12 +++-- cmake/scripts/gen_bazel_build_file.cmake | 44 ++++++++++++++++--- 7 files changed, 72 insertions(+), 13 deletions(-) diff --git a/.github/CONTRIBUTING.md b/.github/CONTRIBUTING.md index 68f82c474..4125b066a 100644 --- a/.github/CONTRIBUTING.md +++ b/.github/CONTRIBUTING.md @@ -158,6 +158,15 @@ make amalgamate Running `make amalgamate` will also apply automatic formatting to the source files using [`Artistic Style`](https://astyle.sourceforge.net/). This formatting may modify your source files in-place. Be certain to review and commit any changes to avoid unintended formatting diffs in commits. +If you add, rename, or remove a header in `include/nlohmann`, also regenerate the header list in +[`BUILD.bazel`](https://github.com/nlohmann/json/blob/develop/BUILD.bazel) (requires CMake) by executing: + +```shell +make BUILD.bazel +``` + +The amalgamation check in CI fails if any of these generated files is out of date. + ## Recommended documentation - The library’s [README file](https://github.com/nlohmann/json/blob/master/README.md) is an excellent starting point to diff --git a/.github/workflows/check_amalgamation.yml b/.github/workflows/check_amalgamation.yml index f70ebfba0..f692e434a 100644 --- a/.github/workflows/check_amalgamation.yml +++ b/.github/workflows/check_amalgamation.yml @@ -57,13 +57,16 @@ jobs: python3 -mvenv venv venv/bin/pip3 install -r $MAIN_DIR/tools/astyle/requirements.txt - - name: Regenerate amalgamation and formatting + - name: Regenerate amalgamation, formatting, and BUILD.bazel run: | cd $MAIN_DIR python3 $TOOL_DIR/amalgamate.py -c $TOOL_DIR/config_json.json -s . python3 $TOOL_DIR/amalgamate.py -c $TOOL_DIR/config_json_fwd.json -s . + # the header list of the Bazel "json" target must match the files in include/ + cmake -P cmake/scripts/gen_bazel_build_file.cmake + ${{ github.workspace }}/venv/bin/astyle --project=tools/astyle/.astylerc --suffix=none --quiet \ $INCLUDE_DIR/json.hpp $INCLUDE_DIR/json_fwd.hpp @@ -87,7 +90,7 @@ jobs: mkdir -p ${{ github.workspace }}/patch git diff --patch --no-color > ${{ github.workspace }}/patch/amalgamation.patch if [ -s ${{ github.workspace }}/patch/amalgamation.patch ]; then - echo "The source code has not been amalgamated/formatted correctly. Diff:" + echo "The source code has not been amalgamated/formatted correctly or BUILD.bazel is out of date. Diff:" cat ${{ github.workspace }}/patch/amalgamation.patch echo "has_diff=true" >> "$GITHUB_OUTPUT" else diff --git a/.github/workflows/comment_check_amalgamation.yml b/.github/workflows/comment_check_amalgamation.yml index 788c1b8ce..4667329d2 100644 --- a/.github/workflows/comment_check_amalgamation.yml +++ b/.github/workflows/comment_check_amalgamation.yml @@ -95,13 +95,13 @@ jobs: issue_number: issue_number, owner: context.repo.owner, repo: context.repo.repo, - body: '## 🔴 Amalgamation check failed! 🔴\nThe source code has not been amalgamated and/or formatted correctly.' + body: '## 🔴 Amalgamation check failed! 🔴\nThe source code has not been amalgamated and/or formatted correctly, or `BUILD.bazel` is out of date.' + (hasPatch ? '\n\n📎 A ready-to-apply patch is attached to the [failed workflow run](' + runUrl + ') as the `amalgamation-patch` artifact.' + ' Download it, then apply it locally from the repository root with:' + '\n\n```shell\ngit apply amalgamation.patch\n```\n\n' + 'This does not require installing astyle yourself.' : '') + (first ? '\n\n@' + author + ' Please read and follow the [Contribution Guidelines]' - + '(https://github.com/nlohmann/json/blob/develop/.github/CONTRIBUTING.md#files-to-change).' + + '(https://github.com/nlohmann/json/blob/develop/.github/CONTRIBUTING.md#amalgamate-the-source-code).' : '') }) diff --git a/BUILD.bazel b/BUILD.bazel index de0ff7145..ea8ffae21 100644 --- a/BUILD.bazel +++ b/BUILD.bazel @@ -30,8 +30,10 @@ cc_library( "include/nlohmann/detail/input/input_adapters.hpp", "include/nlohmann/detail/input/json_sax.hpp", "include/nlohmann/detail/input/lexer.hpp", + "include/nlohmann/detail/input/number_parse.hpp", "include/nlohmann/detail/input/parser.hpp", "include/nlohmann/detail/input/position_t.hpp", + "include/nlohmann/detail/input/string_scan.hpp", "include/nlohmann/detail/iterators/internal_iterator.hpp", "include/nlohmann/detail/iterators/iter_impl.hpp", "include/nlohmann/detail/iterators/iteration_proxy.hpp", @@ -49,6 +51,7 @@ cc_library( "include/nlohmann/detail/meta/detected.hpp", "include/nlohmann/detail/meta/identity_tag.hpp", "include/nlohmann/detail/meta/is_sax.hpp", + "include/nlohmann/detail/meta/logic.hpp", "include/nlohmann/detail/meta/std_fs.hpp", "include/nlohmann/detail/meta/type_traits.hpp", "include/nlohmann/detail/meta/void_t.hpp", diff --git a/FILES.md b/FILES.md index b68167336..263647146 100644 --- a/FILES.md +++ b/FILES.md @@ -250,12 +250,16 @@ Further documentation: ### `BUILD.bazel` -The file can be updated by calling +The build definition for [Bazel](https://bazel.build). The file is generated by +`cmake/scripts/gen_bazel_build_file.cmake`, which derives the header list from the files in `include`; change the +script rather than editing the file by hand. The file can be updated by calling ```shell make BUILD.bazel ``` +The "Check amalgamation" workflow fails if the file is out of date. + ### `meson.build` The build definition for the [Meson](https://mesonbuild.com) build system. diff --git a/Makefile b/Makefile index e1a1d2b75..871ea7995 100644 --- a/Makefile +++ b/Makefile @@ -1,4 +1,4 @@ -.PHONY: pretty clean ChangeLog.md release update_hedley update_hedley_undef +.PHONY: pretty clean ChangeLog.md release update_hedley update_hedley_undef BUILD.bazel ########################################################################## # configuration @@ -30,8 +30,9 @@ AMALGAMATED_FWD_FILE=single_include/nlohmann/json_fwd.hpp # main target all: @echo "amalgamate - amalgamate files single_include/nlohmann/json{,_fwd}.hpp from the include/nlohmann sources" + @echo "BUILD.bazel - regenerate the Bazel BUILD file from the include/nlohmann sources" @echo "ChangeLog.md - generate ChangeLog file" - @echo "check-amalgamation - check whether sources have been amalgamated" + @echo "check-amalgamation - check whether sources have been amalgamated and BUILD.bazel is up to date" @echo "clean - remove built files" @echo "doctest - compile example files and check their output" @echo "fuzz_testing - prepare fuzz testing of the JSON parser" @@ -172,8 +173,13 @@ check-amalgamation: @diff $(AMALGAMATED_FWD_FILE) $(AMALGAMATED_FWD_FILE)~ || (echo "===================================================================\n Amalgamation required! Please read the contribution guidelines\n in file .github/CONTRIBUTING.md.\n===================================================================" ; mv $(AMALGAMATED_FWD_FILE)~ $(AMALGAMATED_FWD_FILE) ; false) @mv $(AMALGAMATED_FILE)~ $(AMALGAMATED_FILE) @mv $(AMALGAMATED_FWD_FILE)~ $(AMALGAMATED_FWD_FILE) + @mv BUILD.bazel BUILD.bazel~ + @$(MAKE) BUILD.bazel + @diff BUILD.bazel BUILD.bazel~ || (echo "===================================================================\n BUILD.bazel is out of date! Please run 'make BUILD.bazel'.\n===================================================================" ; mv BUILD.bazel~ BUILD.bazel ; false) + @mv BUILD.bazel~ BUILD.bazel -BUILD.bazel: $(SRCS) +# generate the Bazel BUILD file; phony, because a removed header would not trigger a rebuild +BUILD.bazel: cmake -P cmake/scripts/gen_bazel_build_file.cmake ########################################################################## diff --git a/cmake/scripts/gen_bazel_build_file.cmake b/cmake/scripts/gen_bazel_build_file.cmake index e754d387d..3c7db9493 100644 --- a/cmake/scripts/gen_bazel_build_file.cmake +++ b/cmake/scripts/gen_bazel_build_file.cmake @@ -1,24 +1,58 @@ # generate Bazel BUILD file +# +# usage: cmake -P cmake/scripts/gen_bazel_build_file.cmake (or: make BUILD.bazel) +# +# The header list of the "json" target is derived from the files in include/. Everything else is fixed text below, +# so edit this script rather than BUILD.bazel. -set(PROJECT_ROOT "${CMAKE_CURRENT_LIST_DIR}/../..") +get_filename_component(PROJECT_ROOT "${CMAKE_CURRENT_LIST_DIR}/../.." ABSOLUTE) set(BUILD_FILE "${PROJECT_ROOT}/BUILD.bazel") -file(GLOB_RECURSE HEADERS LIST_DIRECTORIES false RELATIVE "${PROJECT_ROOT}" "include/*.hpp") +file(GLOB_RECURSE HEADERS LIST_DIRECTORIES false RELATIVE "${PROJECT_ROOT}" "${PROJECT_ROOT}/include/*.hpp") +list(SORT HEADERS) + +set(CONTENT [=[ +load("@rules_cc//cc:cc_library.bzl", "cc_library") +load("@rules_license//rules:license.bzl", "license") + +package( + default_applicable_licenses = [":license"], +) + +exports_files([ + "LICENSE.MIT", +]) + +license( + name = "license", + license_kinds = ["@rules_license//licenses/spdx:MIT"], + license_text = "LICENSE.MIT", +) -file(WRITE "${BUILD_FILE}" [=[ cc_library( name = "json", hdrs = [ ]=]) foreach(header ${HEADERS}) - file(APPEND "${BUILD_FILE}" " \"${header}\",\n") + string(APPEND CONTENT " \"${header}\",\n") endforeach() -file(APPEND "${BUILD_FILE}" [=[ +string(APPEND CONTENT [=[ ], includes = ["include"], visibility = ["//visibility:public"], alwayslink = True, ) + +cc_library( + name = "singleheader-json", + hdrs = [ + "single_include/nlohmann/json.hpp", + ], + includes = ["single_include"], + visibility = ["//visibility:public"], +) ]=]) + +file(WRITE "${BUILD_FILE}" "${CONTENT}") From 7c90ec2323fd85d8067961fb1ab808c6a0d1bac2 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Thu, 24 Sep 2026 17:12:02 +0200 Subject: [PATCH 10/17] Hash deeply nested values without recursing per nesting level (#5546) * Hash deeply nested values without recursing per nesting level std::hash hashed an array or object by hashing each element, which called detail::hash again once per nesting level. A value nested deeply enough - 50,000 levels of objects on an 8 MiB stack - exhausted the call stack and terminated the process. parse() accepts such values without complaint, since the parser is iterative, and a parsed value is hashed wherever it is used as a key in an unordered container. Bound the descent the same way dump() does: detail::hash takes the nesting level, and once hash_depth_limit() (128) levels have been entered, hash_iteratively() hashes what is left on an explicit stack. It combines the seeds in exactly the same order, so hash values are unchanged. A value nested less deeply than the bound is hashed by the same code as before, without allocating, and is as fast as before. Tests check that every depth up to twice the bound hashes exactly like the recursive definition of the hash, and that values nested 100,000 levels deep hash without crashing. Fixes #5545 for std::hash. Signed-off-by: Niels Lohmann * Declare hash_frame's constructor noexcept GCC's -Wnoexcept (an error in CI) flags the emplace_back() into the hash stack under C++26: the constructor cannot throw, since cbegin() is noexcept, but it did not say so. dump_frame's constructor is noexcept for the same reason. Signed-off-by: Niels Lohmann * Share one recursion depth limit, and copy the hash frame out of the stack dump() and hash() each defined their own limit on how many nesting levels they recurse into, and the operations still to come would have added more, free to diverge over time. They now all use detail::recursion_depth_limit(), in a header of its own; serializer::dump_depth_limit() and hash_depth_limit() are gone. hash_iteratively() now copies the frame it works on out of the stack and changes the frame only through stack.back(), so nothing can refer into the stack after entering an element has grown it. Signed-off-by: Niels Lohmann * Parenthesize multiplications in the hash test for clang-tidy Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- BUILD.bazel | 1 + include/nlohmann/detail/hash.hpp | 102 ++++++++++- include/nlohmann/detail/output/serializer.hpp | 16 +- .../nlohmann/detail/recursion_depth_limit.hpp | 35 ++++ include/nlohmann/json.hpp | 1 + single_include/nlohmann/json.hpp | 158 ++++++++++++++++-- tests/src/unit-hash.cpp | 113 +++++++++++++ 7 files changed, 398 insertions(+), 28 deletions(-) create mode 100644 include/nlohmann/detail/recursion_depth_limit.hpp diff --git a/BUILD.bazel b/BUILD.bazel index ea8ffae21..b13e62c22 100644 --- a/BUILD.bazel +++ b/BUILD.bazel @@ -58,6 +58,7 @@ cc_library( "include/nlohmann/detail/output/binary_writer.hpp", "include/nlohmann/detail/output/output_adapters.hpp", "include/nlohmann/detail/output/serializer.hpp", + "include/nlohmann/detail/recursion_depth_limit.hpp", "include/nlohmann/detail/string_concat.hpp", "include/nlohmann/detail/string_escape.hpp", "include/nlohmann/detail/string_utils.hpp", diff --git a/include/nlohmann/detail/hash.hpp b/include/nlohmann/detail/hash.hpp index be8063f89..20a971886 100644 --- a/include/nlohmann/detail/hash.hpp +++ b/include/nlohmann/detail/hash.hpp @@ -11,8 +11,10 @@ #include // uint8_t #include // size_t #include // hash +#include // vector #include +#include #include NLOHMANN_JSON_NAMESPACE_BEGIN @@ -26,6 +28,9 @@ inline std::size_t combine(std::size_t seed, std::size_t h) noexcept return seed; } +template +std::size_t hash_iteratively(const BasicJsonType& j); + /*! @brief hash a JSON value @@ -33,12 +38,21 @@ The hash function tries to rely on std::hash where possible. Furthermore, the type of the JSON value is taken into account to have different hash values for null, 0, 0U, and false, etc. +Hashing an array or an object hashes its elements, which used to call this +function again once per nesting level, so a value nested deeply enough +exhausted the call stack and terminated the process. The descent is bounded +here: once @ref recursion_depth_limit levels have been entered, @ref +hash_iteratively hashes what is left without the call stack. A value nested +less deeply than that - all but a vanishing minority - is hashed exactly as +before, without allocating. + @tparam BasicJsonType basic_json specialization @param j JSON value to hash +@param depth nesting level of @a j, counted from the value passed by the caller @return hash value of j */ template -std::size_t hash(const BasicJsonType& j) +std::size_t hash(const BasicJsonType& j, const std::size_t depth = 0) { using string_t = typename BasicJsonType::string_t; using number_integer_t = typename BasicJsonType::number_integer_t; @@ -56,22 +70,32 @@ std::size_t hash(const BasicJsonType& j) case BasicJsonType::value_t::object: { + if (JSON_HEDLEY_UNLIKELY(depth >= recursion_depth_limit())) + { + return hash_iteratively(j); + } + auto seed = combine(type, j.size()); for (const auto& element : j.items()) { const auto h = std::hash {}(element.key()); seed = combine(seed, h); - seed = combine(seed, hash(element.value())); + seed = combine(seed, hash(element.value(), depth + 1)); } return seed; } case BasicJsonType::value_t::array: { + if (JSON_HEDLEY_UNLIKELY(depth >= recursion_depth_limit())) + { + return hash_iteratively(j); + } + auto seed = combine(type, j.size()); for (const auto& element : j) { - seed = combine(seed, hash(element)); + seed = combine(seed, hash(element, depth + 1)); } return seed; } @@ -127,5 +151,77 @@ std::size_t hash(const BasicJsonType& j) } } +/// an array or object whose elements @ref hash_iteratively is hashing +template +struct hash_frame +{ + hash_frame(const BasicJsonType* value_, std::size_t seed_) noexcept + : value(value_), position(value_->cbegin()), seed(seed_) + {} + + const BasicJsonType* value; + typename BasicJsonType::const_iterator position; + std::size_t seed; +}; + +/*! +@brief hash the array or object @a j without the call stack + +Computes the same value as @ref hash, keeping the arrays and objects it has +entered on an explicit stack instead of descending into them. Only reached for +values nested deeper than @ref recursion_depth_limit. + +@tparam BasicJsonType basic_json specialization +@param j array or object to hash +@return hash value of j +*/ +template +std::size_t hash_iteratively(const BasicJsonType& j) +{ + using string_t = typename BasicJsonType::string_t; + + std::vector> stack; + stack.emplace_back(&j, combine(static_cast(j.type()), j.size())); + + while (true) + { + // a copy, as entering an element below can reallocate the stack; the + // frame itself is only changed through stack.back() + const hash_frame frame = stack.back(); + + if (frame.position == frame.value->cend()) + { + // all elements are hashed: fold this value's hash into its parent's + // seed, exactly where the recursive version returns it + const std::size_t h = frame.seed; + stack.pop_back(); + if (stack.empty()) + { + return h; + } + stack.back().seed = combine(stack.back().seed, h); + continue; + } + + if (frame.value->is_object()) + { + stack.back().seed = combine(stack.back().seed, std::hash {}(frame.position.key())); + } + + // advance before entering the element, which pushes onto the stack + const BasicJsonType& element = *frame.position; + ++stack.back().position; + + if (element.is_structured()) + { + stack.emplace_back(&element, combine(static_cast(element.type()), element.size())); + } + else + { + stack.back().seed = combine(stack.back().seed, hash(element)); + } + } +} + } // namespace detail NLOHMANN_JSON_NAMESPACE_END diff --git a/include/nlohmann/detail/output/serializer.hpp b/include/nlohmann/detail/output/serializer.hpp index f9e7f7840..7c38276ce 100644 --- a/include/nlohmann/detail/output/serializer.hpp +++ b/include/nlohmann/detail/output/serializer.hpp @@ -30,6 +30,7 @@ #include #include #include +#include #include #include @@ -133,7 +134,7 @@ class serializer Serializing a container descends into its elements, so a value nested deeply enough used to exhaust the call stack and terminate the process with no - exception to catch. The descent is bounded here: once @ref dump_depth_limit + exception to catch. The descent is bounded here: once @ref recursion_depth_limit levels have been entered, @ref dump_iteratively writes out what is left without the call stack. A value nested less deeply than that - all but a vanishing minority - is written by exactly the code that always wrote it. @@ -148,7 +149,7 @@ class serializer { case value_t::object: { - if (JSON_HEDLEY_UNLIKELY(depth >= dump_depth_limit())) + if (JSON_HEDLEY_UNLIKELY(depth >= recursion_depth_limit())) { dump_iteratively(val, current_indent); return; @@ -223,7 +224,7 @@ class serializer case value_t::array: { - if (JSON_HEDLEY_UNLIKELY(depth >= dump_depth_limit())) + if (JSON_HEDLEY_UNLIKELY(depth >= recursion_depth_limit())) { dump_iteratively(val, current_indent); return; @@ -408,19 +409,12 @@ class serializer } private: - /// the number of levels @ref dump_internal descends into before it hands - /// over to @ref dump_iteratively - static constexpr std::size_t dump_depth_limit() - { - return 128; - } - /*! @brief write out @a val and everything below it without the call stack Emits the same bytes as @ref dump_internal, keeping the containers it has entered on an explicit stack instead of descending into them. Only reached - for values nested deeper than @ref dump_depth_limit, which is why it is not + for values nested deeper than @ref recursion_depth_limit, which is why it is not written for speed: walking every value this way measured up to 20% slower on object-heavy documents than letting the compiler drive the descent. */ diff --git a/include/nlohmann/detail/recursion_depth_limit.hpp b/include/nlohmann/detail/recursion_depth_limit.hpp new file mode 100644 index 000000000..fe3bd8026 --- /dev/null +++ b/include/nlohmann/detail/recursion_depth_limit.hpp @@ -0,0 +1,35 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +#include // size_t + +#include + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ + +/*! +@brief the number of nesting levels an operation recurses into + +Operations that walk a value (serializing, hashing, merging, ...) recurse once +per nesting level, which is fastest, but a value nested deeply enough would +exhaust the call stack. So they recurse only this many levels deep and finish +whatever lies below with an explicit stack. All of them share this limit. + +@sa https://github.com/nlohmann/json/issues/5387 +*/ +constexpr std::size_t recursion_depth_limit() noexcept +{ + return 128; +} + +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END diff --git a/include/nlohmann/json.hpp b/include/nlohmann/json.hpp index 9c1d821e1..b64feb4a9 100644 --- a/include/nlohmann/json.hpp +++ b/include/nlohmann/json.hpp @@ -68,6 +68,7 @@ #include #include #include +#include #include #include #include diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index c11fa27e5..335a29289 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -7028,9 +7028,48 @@ NLOHMANN_JSON_NAMESPACE_END #include // uint8_t #include // size_t #include // hash +#include // vector // #include +// #include +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + + + +#include // size_t + +// #include + + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ + +/*! +@brief the number of nesting levels an operation recurses into + +Operations that walk a value (serializing, hashing, merging, ...) recurse once +per nesting level, which is fastest, but a value nested deeply enough would +exhaust the call stack. So they recurse only this many levels deep and finish +whatever lies below with an explicit stack. All of them share this limit. + +@sa https://github.com/nlohmann/json/issues/5387 +*/ +constexpr std::size_t recursion_depth_limit() noexcept +{ + return 128; +} + +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END + // #include @@ -7045,6 +7084,9 @@ inline std::size_t combine(std::size_t seed, std::size_t h) noexcept return seed; } +template +std::size_t hash_iteratively(const BasicJsonType& j); + /*! @brief hash a JSON value @@ -7052,12 +7094,21 @@ The hash function tries to rely on std::hash where possible. Furthermore, the type of the JSON value is taken into account to have different hash values for null, 0, 0U, and false, etc. +Hashing an array or an object hashes its elements, which used to call this +function again once per nesting level, so a value nested deeply enough +exhausted the call stack and terminated the process. The descent is bounded +here: once @ref recursion_depth_limit levels have been entered, @ref +hash_iteratively hashes what is left without the call stack. A value nested +less deeply than that - all but a vanishing minority - is hashed exactly as +before, without allocating. + @tparam BasicJsonType basic_json specialization @param j JSON value to hash +@param depth nesting level of @a j, counted from the value passed by the caller @return hash value of j */ template -std::size_t hash(const BasicJsonType& j) +std::size_t hash(const BasicJsonType& j, const std::size_t depth = 0) { using string_t = typename BasicJsonType::string_t; using number_integer_t = typename BasicJsonType::number_integer_t; @@ -7075,22 +7126,32 @@ std::size_t hash(const BasicJsonType& j) case BasicJsonType::value_t::object: { + if (JSON_HEDLEY_UNLIKELY(depth >= recursion_depth_limit())) + { + return hash_iteratively(j); + } + auto seed = combine(type, j.size()); for (const auto& element : j.items()) { const auto h = std::hash {}(element.key()); seed = combine(seed, h); - seed = combine(seed, hash(element.value())); + seed = combine(seed, hash(element.value(), depth + 1)); } return seed; } case BasicJsonType::value_t::array: { + if (JSON_HEDLEY_UNLIKELY(depth >= recursion_depth_limit())) + { + return hash_iteratively(j); + } + auto seed = combine(type, j.size()); for (const auto& element : j) { - seed = combine(seed, hash(element)); + seed = combine(seed, hash(element, depth + 1)); } return seed; } @@ -7146,6 +7207,78 @@ std::size_t hash(const BasicJsonType& j) } } +/// an array or object whose elements @ref hash_iteratively is hashing +template +struct hash_frame +{ + hash_frame(const BasicJsonType* value_, std::size_t seed_) noexcept + : value(value_), position(value_->cbegin()), seed(seed_) + {} + + const BasicJsonType* value; + typename BasicJsonType::const_iterator position; + std::size_t seed; +}; + +/*! +@brief hash the array or object @a j without the call stack + +Computes the same value as @ref hash, keeping the arrays and objects it has +entered on an explicit stack instead of descending into them. Only reached for +values nested deeper than @ref recursion_depth_limit. + +@tparam BasicJsonType basic_json specialization +@param j array or object to hash +@return hash value of j +*/ +template +std::size_t hash_iteratively(const BasicJsonType& j) +{ + using string_t = typename BasicJsonType::string_t; + + std::vector> stack; + stack.emplace_back(&j, combine(static_cast(j.type()), j.size())); + + while (true) + { + // a copy, as entering an element below can reallocate the stack; the + // frame itself is only changed through stack.back() + const hash_frame frame = stack.back(); + + if (frame.position == frame.value->cend()) + { + // all elements are hashed: fold this value's hash into its parent's + // seed, exactly where the recursive version returns it + const std::size_t h = frame.seed; + stack.pop_back(); + if (stack.empty()) + { + return h; + } + stack.back().seed = combine(stack.back().seed, h); + continue; + } + + if (frame.value->is_object()) + { + stack.back().seed = combine(stack.back().seed, std::hash {}(frame.position.key())); + } + + // advance before entering the element, which pushes onto the stack + const BasicJsonType& element = *frame.position; + ++stack.back().position; + + if (element.is_structured()) + { + stack.emplace_back(&element, combine(static_cast(element.type()), element.size())); + } + else + { + stack.back().seed = combine(stack.back().seed, hash(element)); + } + } +} + } // namespace detail NLOHMANN_JSON_NAMESPACE_END @@ -22295,6 +22428,8 @@ NLOHMANN_JSON_NAMESPACE_END // #include +// #include + // #include // #include @@ -22400,7 +22535,7 @@ class serializer Serializing a container descends into its elements, so a value nested deeply enough used to exhaust the call stack and terminate the process with no - exception to catch. The descent is bounded here: once @ref dump_depth_limit + exception to catch. The descent is bounded here: once @ref recursion_depth_limit levels have been entered, @ref dump_iteratively writes out what is left without the call stack. A value nested less deeply than that - all but a vanishing minority - is written by exactly the code that always wrote it. @@ -22415,7 +22550,7 @@ class serializer { case value_t::object: { - if (JSON_HEDLEY_UNLIKELY(depth >= dump_depth_limit())) + if (JSON_HEDLEY_UNLIKELY(depth >= recursion_depth_limit())) { dump_iteratively(val, current_indent); return; @@ -22490,7 +22625,7 @@ class serializer case value_t::array: { - if (JSON_HEDLEY_UNLIKELY(depth >= dump_depth_limit())) + if (JSON_HEDLEY_UNLIKELY(depth >= recursion_depth_limit())) { dump_iteratively(val, current_indent); return; @@ -22675,19 +22810,12 @@ class serializer } private: - /// the number of levels @ref dump_internal descends into before it hands - /// over to @ref dump_iteratively - static constexpr std::size_t dump_depth_limit() - { - return 128; - } - /*! @brief write out @a val and everything below it without the call stack Emits the same bytes as @ref dump_internal, keeping the containers it has entered on an explicit stack instead of descending into them. Only reached - for values nested deeper than @ref dump_depth_limit, which is why it is not + for values nested deeper than @ref recursion_depth_limit, which is why it is not written for speed: walking every value this way measured up to 20% slower on object-heavy documents than letting the compiler drive the descent. */ @@ -24000,6 +24128,8 @@ class serializer } // namespace detail NLOHMANN_JSON_NAMESPACE_END +// #include + // #include // #include diff --git a/tests/src/unit-hash.cpp b/tests/src/unit-hash.cpp index c161efa6e..eb843c291 100644 --- a/tests/src/unit-hash.cpp +++ b/tests/src/unit-hash.cpp @@ -13,6 +13,78 @@ using json = nlohmann::json; using ordered_json = nlohmann::ordered_json; #include +#include + +namespace +{ +// how detail::hash defines the hash of an array or object: the seeds of the +// elements, combined in order. Recursive, so only usable on values nested a +// few hundred levels deep - which is exactly what is needed to check that the +// iterative path taken below detail::recursion_depth_limit() computes the same. +template +std::size_t reference_hash(const BasicJsonType& j) +{ + using nlohmann::detail::combine; + using string_t = typename BasicJsonType::string_t; + + if (!j.is_structured()) + { + return std::hash {}(j); + } + + auto seed = combine(static_cast(j.type()), j.size()); + for (const auto& element : j.items()) + { + if (j.is_object()) + { + seed = combine(seed, std::hash {}(element.key())); + } + seed = combine(seed, reference_hash(element.value())); + } + return seed; +} + +// a value nested `depth` levels deep, with siblings on every level +template +BasicJsonType nested(const std::size_t depth, const bool objects) +{ + BasicJsonType value = "leaf"; + for (std::size_t i = 0; i < depth; ++i) + { + if (objects) + { + value = BasicJsonType{{"before", i}, {"nested", std::move(value)}, {"after", {i, "x"}}}; + } + else + { + value = BasicJsonType::array({i, std::move(value), BasicJsonType::object({{"k", i}})}); + } + } + return value; +} + +std::string nested_text(const std::size_t depth, const bool objects) +{ + std::string text; + if (objects) + { + text.reserve((6 * depth) + 1); + for (std::size_t i = 0; i < depth; ++i) + { + text += "{\"a\":"; + } + text += "1"; + text.append(depth, '}'); + } + else + { + text.assign(depth, '['); + text += "1"; + text.append(depth, ']'); + } + return text; +} +} // namespace TEST_CASE("hash") { @@ -111,3 +183,44 @@ TEST_CASE("hash") CHECK(hashes.size() == 21); } + +TEST_CASE("hash of deeply nested values") +{ + SECTION("hashing past the descent bound computes the same values") + { + // every depth on either side of where the iterative path takes over + for (std::size_t depth = 0; depth <= (2 * nlohmann::detail::recursion_depth_limit()) + 10; ++depth) + { + CAPTURE(depth); + const auto arrays = nested(depth, false); + const auto objects = nested(depth, true); + const auto ordered = nested(depth, true); + CHECK(std::hash {}(arrays) == reference_hash(arrays)); + CHECK(std::hash {}(objects) == reference_hash(objects)); + CHECK(std::hash {}(ordered) == reference_hash(ordered)); + } + } + + SECTION("values nested too deeply for the call stack (#5545)") + { + // recursing once per level used to exhaust the call stack here; the + // values are only parsed and hashed, never copied or compared, since + // those recurse as well + const std::size_t depth = 100000; + for (const bool objects : + { + false, true + }) + { + CAPTURE(objects); + const auto text = nested_text(depth, objects); + const auto a = json::parse(text); + const auto b = json::parse(text); + CHECK(std::hash {}(a) == std::hash {}(b)); + + const auto c = ordered_json::parse(text); + const auto d = ordered_json::parse(text); + CHECK(std::hash {}(c) == std::hash {}(d)); + } + } +} From 4daca40d7b03d05d1af730003f826c1ff91a90e8 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Thu, 24 Sep 2026 17:12:03 +0200 Subject: [PATCH 11/17] Merge deeply nested objects without recursing per nesting level (#5547) * Merge deeply nested objects without recursing per nesting level merge_patch() and update(j, true) merged a nested object by calling themselves on it, once per nesting level. A value nested deeply enough - 50,000 levels of objects on an 8 MiB stack - exhausted the call stack and terminated the process, although parse() accepts such values without complaint. Bound the descent the same way dump() does. The recursion now carries the nesting level, and once merge_depth_limit() (128) levels have been entered, update_members_iteratively() and merge_patch_iteratively() finish the merge on an explicit stack. They still merge a nested object completely before the next member, and in the same order, so the results, including the parents JSON_DIAGNOSTICS reports paths from, are unchanged. Values nested less deeply than the bound run the same code as before, so the common case does not pay for the stack: merging only on it cost 10-14% in a first version. The public signatures are unchanged. The recursive worker behind merge_patch() has its own name rather than being a private overload, so that &basic_json::merge_patch stays unambiguous. Tests check every depth up to 300 against recursive reference implementations of both operations, check the diagnostic paths past the bound, and merge objects nested 100,000 levels deep. Fixes #5545 for update(j, true), and #5393 for merge_patch(). Signed-off-by: Niels Lohmann * Use the shared recursion limit in update() and merge_patch() merge_depth_limit() is gone in favor of detail::recursion_depth_limit(). The two identical function-local frame structs become one member struct, merge_frame, with a constructor, so both loops emplace_back() their frames. merge_patch_iteratively() copies the frame it works on out of the stack and changes it only through stack.back(). Signed-off-by: Niels Lohmann * Build the update()/merge_patch() diagnostics test values instead of parsing them Parsed values carry byte positions under JSON_DIAGNOSTIC_POSITIONS, which the expected messages do not include. Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- include/nlohmann/json.hpp | 178 ++++++++++++++++++++++++++++++- single_include/nlohmann/json.hpp | 178 ++++++++++++++++++++++++++++++- tests/src/unit-diagnostics.cpp | 49 +++++++++ tests/src/unit-merge_patch.cpp | 103 ++++++++++++++++++ tests/src/unit-modifiers.cpp | 88 +++++++++++++++ 5 files changed, 590 insertions(+), 6 deletions(-) diff --git a/include/nlohmann/json.hpp b/include/nlohmann/json.hpp index b64feb4a9..e40a2aa27 100644 --- a/include/nlohmann/json.hpp +++ b/include/nlohmann/json.hpp @@ -3913,17 +3913,54 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_THROW(type_error::create(312, detail::concat("cannot use update() with ", first.m_object->type_name()), first.m_object)); } + update_members(first, last, merge_objects, 0); + } + + private: + /// @brief an object @ref update_members_iteratively or @ref + /// merge_patch_iteratively is merging into, and the members still to merge + struct merge_frame + { + merge_frame(basic_json* target_, const_iterator position_, const_iterator last_) noexcept + : target(target_), position(std::move(position_)), last(std::move(last_)) + {} + + basic_json* target; + const_iterator position; + const_iterator last; + }; + + /*! + @brief the members loop of @ref update, for this object and range + + Merging a nested object calls this function again, once per nesting + level, so a value nested deeply enough used to exhaust the call stack and + terminate the process. The descent is bounded here: once @ref + detail::recursion_depth_limit levels have been entered, @ref + update_members_iteratively merges what is left without the call stack. + + @param[in] depth nesting level of this object, counted from the object + @ref update was called on + */ + void update_members(const const_iterator& first, const const_iterator& last, const bool merge_objects, const std::size_t depth) + { + if (JSON_HEDLEY_UNLIKELY(depth >= detail::recursion_depth_limit())) + { + update_members_iteratively(first, last); + return; + } + for (auto it = first; it != last; ++it) { if (merge_objects && it.value().is_object()) { - auto it2 = m_data.m_value.object->find(it.key()); + const auto it2 = m_data.m_value.object->find(it.key()); // Only recurse when the existing value is itself an object. // Otherwise overwrite, matching the documented "all other values // are overwritten as usual" behavior (see #5402). if (it2 != m_data.m_value.object->end() && it2->second.is_object()) { - it2->second.update(it.value(), true); + it2->second.update_members(it.value().cbegin(), it.value().cend(), true, depth + 1); #if JSON_DIAGNOSTICS it2->second.set_parents(); #endif @@ -3937,6 +3974,64 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } } + /*! + @brief merge @a first to @a last into this object without the call stack + + Does the same as @ref update_members with `merge_objects` set, keeping the + objects whose merge was interrupted by a nested one on an explicit stack + instead of descending into them. A nested object is still merged + completely before the next member, in the same order as the recursive + version. Only reached for values nested deeper than @ref + detail::recursion_depth_limit. + */ + void update_members_iteratively(const_iterator first, const_iterator last) + { + std::vector stack; + + basic_json* target = this; + while (true) + { + if (first == last) + { + if (stack.empty()) + { + break; + } + + // a nested object is merged: continue with its parent +#if JSON_DIAGNOSTICS + target->set_parents(); +#endif + target = stack.back().target; + first = stack.back().position; + last = stack.back().last; + stack.pop_back(); + continue; + } + + if (first.value().is_object()) + { + const auto it2 = target->m_data.m_value.object->find(first.key()); + if (it2 != target->m_data.m_value.object->end() && it2->second.is_object()) + { + const basic_json& source = first.value(); + ++first; + stack.emplace_back(target, first, last); + target = &it2->second; + first = source.cbegin(); + last = source.cend(); + continue; + } + } + target->m_data.m_value.object->operator[](first.key()) = first.value(); +#if JSON_DIAGNOSTICS + target->m_data.m_value.object->operator[](first.key()).m_parent = target; +#endif + ++first; + } + } + + public: /// @brief exchanges the values /// @sa https://json.nlohmann.me/api/basic_json/swap/ void swap(reference other) noexcept ( @@ -5831,9 +5926,30 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @brief applies a JSON Merge Patch /// @sa https://json.nlohmann.me/api/basic_json/merge_patch/ void merge_patch(const basic_json& apply_patch) + { + apply_merge_patch(apply_patch, 0); + } + + private: + /*! + @brief @ref merge_patch, for a patch at nesting level @a depth + + Applying a nested object calls this function again, once per nesting + level, so a patch nested deeply enough used to exhaust the call stack and + terminate the process. The descent is bounded here: once @ref + detail::recursion_depth_limit levels have been entered, @ref + merge_patch_iteratively applies what is left without the call stack. + */ + void apply_merge_patch(const basic_json& apply_patch, const std::size_t depth) { if (apply_patch.is_object()) { + if (JSON_HEDLEY_UNLIKELY(depth >= detail::recursion_depth_limit())) + { + merge_patch_iteratively(apply_patch); + return; + } + if (!is_object()) { *this = object(); @@ -5846,7 +5962,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } else { - operator[](it.key()).merge_patch(it.value()); + operator[](it.key()).apply_merge_patch(it.value(), depth + 1); } } } @@ -5856,6 +5972,62 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } } + /*! + @brief apply @a apply_patch to this value without the call stack + + Does the same as @ref merge_patch, keeping the objects being patched on an + explicit stack instead of descending into them. A nested object is still + patched completely before the next member, in the same order as the + recursive version. Only reached for patches nested deeper than @ref + detail::recursion_depth_limit. + */ + void merge_patch_iteratively(const basic_json& apply_patch) + { + std::vector stack; + + // patch `target` with `patch`, or start patching it member by member + const auto apply = [&stack](basic_json & target, const basic_json & patch) + { + if (patch.is_object()) + { + if (!target.is_object()) + { + target = basic_json::object(); + } + stack.emplace_back(&target, patch.cbegin(), patch.cend()); + } + else + { + target = patch; + } + }; + + apply(*this, apply_patch); + while (!stack.empty()) + { + // a copy, as applying a member below can reallocate the stack; + // the frame itself is only changed through stack.back() + const merge_frame frame = stack.back(); + if (frame.position == frame.last) + { + stack.pop_back(); + continue; + } + + const const_iterator member = frame.position; + ++stack.back().position; + if (member.value().is_null()) + { + frame.target->erase(member.key()); + } + else + { + apply(frame.target->operator[](member.key()), member.value()); + } + } + } + + public: /// @} }; diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 335a29289..3c5821d92 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -28371,17 +28371,54 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_THROW(type_error::create(312, detail::concat("cannot use update() with ", first.m_object->type_name()), first.m_object)); } + update_members(first, last, merge_objects, 0); + } + + private: + /// @brief an object @ref update_members_iteratively or @ref + /// merge_patch_iteratively is merging into, and the members still to merge + struct merge_frame + { + merge_frame(basic_json* target_, const_iterator position_, const_iterator last_) noexcept + : target(target_), position(std::move(position_)), last(std::move(last_)) + {} + + basic_json* target; + const_iterator position; + const_iterator last; + }; + + /*! + @brief the members loop of @ref update, for this object and range + + Merging a nested object calls this function again, once per nesting + level, so a value nested deeply enough used to exhaust the call stack and + terminate the process. The descent is bounded here: once @ref + detail::recursion_depth_limit levels have been entered, @ref + update_members_iteratively merges what is left without the call stack. + + @param[in] depth nesting level of this object, counted from the object + @ref update was called on + */ + void update_members(const const_iterator& first, const const_iterator& last, const bool merge_objects, const std::size_t depth) + { + if (JSON_HEDLEY_UNLIKELY(depth >= detail::recursion_depth_limit())) + { + update_members_iteratively(first, last); + return; + } + for (auto it = first; it != last; ++it) { if (merge_objects && it.value().is_object()) { - auto it2 = m_data.m_value.object->find(it.key()); + const auto it2 = m_data.m_value.object->find(it.key()); // Only recurse when the existing value is itself an object. // Otherwise overwrite, matching the documented "all other values // are overwritten as usual" behavior (see #5402). if (it2 != m_data.m_value.object->end() && it2->second.is_object()) { - it2->second.update(it.value(), true); + it2->second.update_members(it.value().cbegin(), it.value().cend(), true, depth + 1); #if JSON_DIAGNOSTICS it2->second.set_parents(); #endif @@ -28395,6 +28432,64 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } } + /*! + @brief merge @a first to @a last into this object without the call stack + + Does the same as @ref update_members with `merge_objects` set, keeping the + objects whose merge was interrupted by a nested one on an explicit stack + instead of descending into them. A nested object is still merged + completely before the next member, in the same order as the recursive + version. Only reached for values nested deeper than @ref + detail::recursion_depth_limit. + */ + void update_members_iteratively(const_iterator first, const_iterator last) + { + std::vector stack; + + basic_json* target = this; + while (true) + { + if (first == last) + { + if (stack.empty()) + { + break; + } + + // a nested object is merged: continue with its parent +#if JSON_DIAGNOSTICS + target->set_parents(); +#endif + target = stack.back().target; + first = stack.back().position; + last = stack.back().last; + stack.pop_back(); + continue; + } + + if (first.value().is_object()) + { + const auto it2 = target->m_data.m_value.object->find(first.key()); + if (it2 != target->m_data.m_value.object->end() && it2->second.is_object()) + { + const basic_json& source = first.value(); + ++first; + stack.emplace_back(target, first, last); + target = &it2->second; + first = source.cbegin(); + last = source.cend(); + continue; + } + } + target->m_data.m_value.object->operator[](first.key()) = first.value(); +#if JSON_DIAGNOSTICS + target->m_data.m_value.object->operator[](first.key()).m_parent = target; +#endif + ++first; + } + } + + public: /// @brief exchanges the values /// @sa https://json.nlohmann.me/api/basic_json/swap/ void swap(reference other) noexcept ( @@ -30289,9 +30384,30 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @brief applies a JSON Merge Patch /// @sa https://json.nlohmann.me/api/basic_json/merge_patch/ void merge_patch(const basic_json& apply_patch) + { + apply_merge_patch(apply_patch, 0); + } + + private: + /*! + @brief @ref merge_patch, for a patch at nesting level @a depth + + Applying a nested object calls this function again, once per nesting + level, so a patch nested deeply enough used to exhaust the call stack and + terminate the process. The descent is bounded here: once @ref + detail::recursion_depth_limit levels have been entered, @ref + merge_patch_iteratively applies what is left without the call stack. + */ + void apply_merge_patch(const basic_json& apply_patch, const std::size_t depth) { if (apply_patch.is_object()) { + if (JSON_HEDLEY_UNLIKELY(depth >= detail::recursion_depth_limit())) + { + merge_patch_iteratively(apply_patch); + return; + } + if (!is_object()) { *this = object(); @@ -30304,7 +30420,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } else { - operator[](it.key()).merge_patch(it.value()); + operator[](it.key()).apply_merge_patch(it.value(), depth + 1); } } } @@ -30314,6 +30430,62 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } } + /*! + @brief apply @a apply_patch to this value without the call stack + + Does the same as @ref merge_patch, keeping the objects being patched on an + explicit stack instead of descending into them. A nested object is still + patched completely before the next member, in the same order as the + recursive version. Only reached for patches nested deeper than @ref + detail::recursion_depth_limit. + */ + void merge_patch_iteratively(const basic_json& apply_patch) + { + std::vector stack; + + // patch `target` with `patch`, or start patching it member by member + const auto apply = [&stack](basic_json & target, const basic_json & patch) + { + if (patch.is_object()) + { + if (!target.is_object()) + { + target = basic_json::object(); + } + stack.emplace_back(&target, patch.cbegin(), patch.cend()); + } + else + { + target = patch; + } + }; + + apply(*this, apply_patch); + while (!stack.empty()) + { + // a copy, as applying a member below can reallocate the stack; + // the frame itself is only changed through stack.back() + const merge_frame frame = stack.back(); + if (frame.position == frame.last) + { + stack.pop_back(); + continue; + } + + const const_iterator member = frame.position; + ++stack.back().position; + if (member.value().is_null()) + { + frame.target->erase(member.key()); + } + else + { + apply(frame.target->operator[](member.key()), member.value()); + } + } + } + + public: /// @} }; diff --git a/tests/src/unit-diagnostics.cpp b/tests/src/unit-diagnostics.cpp index 389a7a3d7..cdb6185b3 100644 --- a/tests/src/unit-diagnostics.cpp +++ b/tests/src/unit-diagnostics.cpp @@ -363,3 +363,52 @@ TEST_CASE("Regression tests for extended diagnostics") } } +TEST_CASE("Better diagnostics past the descent bound of update() and merge_patch()") +{ + // Both merge objects nested more than detail::recursion_depth_limit() + // (128) levels deep without recursing; the values they add or replace + // there must still know their parents. + // The values are built rather than parsed, so that the expected messages + // carry no byte positions under JSON_DIAGNOSTIC_POSITIONS. + const std::size_t depth = 200; + json target = {{"x", 1}}; + json patch = {{"y", 2}}; + std::string path; + for (std::size_t i = 0; i < depth; ++i) + { + target = json{{"a", std::move(target)}}; + patch = json{{"a", std::move(patch)}}; + path += "/a"; + } + const std::string expected_x = "[json.exception.type_error.304] (" + path + "/x) cannot use at() with number"; + const std::string expected_y = "[json.exception.type_error.304] (" + path + "/y) cannot use at() with number"; + + SECTION("update()") + { + json j = target; + j.update(patch, true); + + // walk down through const references, which leave m_parent alone + const json* p = &j; + for (std::size_t i = 0; i < depth; ++i) + { + p = &p->at("a"); + } + CHECK_THROWS_WITH_AS(p->at("x").at(0), expected_x.c_str(), json::type_error); + CHECK_THROWS_WITH_AS(p->at("y").at(0), expected_y.c_str(), json::type_error); + } + + SECTION("merge_patch()") + { + json j = target; + j.merge_patch(patch); + + const json* p = &j; + for (std::size_t i = 0; i < depth; ++i) + { + p = &p->at("a"); + } + CHECK_THROWS_WITH_AS(p->at("x").at(0), expected_x.c_str(), json::type_error); + CHECK_THROWS_WITH_AS(p->at("y").at(0), expected_y.c_str(), json::type_error); + } +} diff --git a/tests/src/unit-merge_patch.cpp b/tests/src/unit-merge_patch.cpp index f02a1e991..8ac281ef5 100644 --- a/tests/src/unit-merge_patch.cpp +++ b/tests/src/unit-merge_patch.cpp @@ -14,6 +14,60 @@ using nlohmann::json; using namespace nlohmann::literals; // NOLINT(google-build-using-namespace) #endif +#include + +namespace +{ +// RFC 7396's MergePatch, written recursively as in the RFC; only usable on +// values nested a few hundred levels deep +void reference_merge_patch(json& target, const json& patch) +{ + if (!patch.is_object()) + { + target = patch; + return; + } + if (!target.is_object()) + { + target = json::object(); + } + for (auto it = patch.begin(); it != patch.end(); ++it) + { + if (it.value().is_null()) + { + target.erase(it.key()); + } + else + { + reference_merge_patch(target[it.key()], it.value()); + } + } +} + +// objects nested `depth` levels deep under the key "a", with members that +// differ by `variant` on the way down +std::string nested_objects(const std::size_t depth, const int variant) +{ + std::string text; + for (std::size_t i = 0; i < depth; ++i) + { + text += "{"; + if ((i + static_cast(variant)) % 3 == 0) + { + text += "\"s" + std::to_string(variant) + "\":" + std::to_string(i) + ","; + } + if (variant == 2 && i % 5 == 0) + { + text += "\"s0\":null,"; + } + text += "\"a\":"; + } + text += variant == 1 ? "{\"x\":1,\"y\":null}" : "{\"y\":2}"; + text.append(depth, '}'); + return text; +} +} // namespace + TEST_CASE("JSON Merge Patch") { SECTION("examples from RFC 7396") @@ -242,3 +296,52 @@ TEST_CASE("JSON Merge Patch") } } } + +TEST_CASE("JSON Merge Patch on deeply nested values") +{ + SECTION("patching past the descent bound gives the same result") + { + // every depth on either side of where the iterative version takes + // over (detail::recursion_depth_limit(), 128) + for (std::size_t depth = 0; depth <= 300; ++depth) + { + CAPTURE(depth); + for (int variant = 0; variant < 3; ++variant) + { + CAPTURE(variant); + const json patch = json::parse(nested_objects(depth, variant)); + + json result = json::parse(nested_objects(depth, (variant + 1) % 3)); + json expected = result; + result.merge_patch(patch); + reference_merge_patch(expected, patch); + CHECK(result == expected); + + // a target that is not an object, and an empty one + json from_null; + from_null.merge_patch(patch); + json expected_from_null; + reference_merge_patch(expected_from_null, patch); + CHECK(from_null == expected_from_null); + } + } + } + + SECTION("patches nested too deeply for the call stack (#5393)") + { + // applying a patch used to recurse once per nesting level. The result + // is only walked, never copied or compared, since those recurse too. + const std::size_t depth = 100000; + json target = json::parse(nested_objects(depth, 0)); + target.merge_patch(json::parse(nested_objects(depth, 1))); + + const json* p = ⌖ + for (std::size_t i = 0; i < depth; ++i) + { + p = &p->at("a"); + } + // {"y":2} patched with {"x":1,"y":null} + CHECK(p->size() == 1); + CHECK(p->at("x") == 1); + } +} diff --git a/tests/src/unit-modifiers.cpp b/tests/src/unit-modifiers.cpp index 369162772..55f9d467e 100644 --- a/tests/src/unit-modifiers.cpp +++ b/tests/src/unit-modifiers.cpp @@ -11,6 +11,53 @@ #include using nlohmann::json; +#include + +namespace +{ +// update(source, true) as documented, written recursively; only usable on +// values nested a few hundred levels deep +void reference_update(json& target, const json& source) +{ + for (auto it = source.begin(); it != source.end(); ++it) + { + const auto existing = target.find(it.key()); + if (it.value().is_object() && existing != target.end() && existing->is_object()) + { + reference_update(*existing, it.value()); + } + else + { + target[it.key()] = it.value(); + } + } +} + +// objects nested `depth` levels deep under the key "a", with members that +// differ by `variant` on the way down +std::string nested_objects(const std::size_t depth, const int variant) +{ + std::string text; + for (std::size_t i = 0; i < depth; ++i) + { + text += "{"; + if ((i + static_cast(variant)) % 3 == 0) + { + text += "\"s" + std::to_string(variant) + "\":" + std::to_string(i) + ","; + } + if (variant == 2 && i % 5 == 0) + { + // an object replacing a primitive, which is not merged + text += "\"s0\":{\"o\":1},"; + } + text += "\"a\":"; + } + text += variant == 1 ? "{\"x\":1}" : "{\"y\":2}"; + text.append(depth, '}'); + return text; +} +} // namespace + TEST_CASE("modifiers") { SECTION("clear()") @@ -988,3 +1035,44 @@ TEST_CASE("modifiers") } } } + +TEST_CASE("update() on deeply nested values") +{ + SECTION("merging past the descent bound gives the same result") + { + // every depth on either side of where the iterative version takes + // over (detail::recursion_depth_limit(), 128) + for (std::size_t depth = 0; depth <= 300; ++depth) + { + CAPTURE(depth); + for (int variant = 0; variant < 3; ++variant) + { + CAPTURE(variant); + const json source = json::parse(nested_objects(depth, variant)); + json result = json::parse(nested_objects(depth, (variant + 1) % 3)); + json expected = result; + result.update(source, true); + reference_update(expected, source); + CHECK(result == expected); + } + } + } + + SECTION("objects nested too deeply for the call stack (#5545)") + { + // merging used to recurse once per nesting level. The result is only + // walked, never copied or compared, since those recurse too. + const std::size_t depth = 100000; + json target = json::parse(nested_objects(depth, 0)); + target.update(json::parse(nested_objects(depth, 1)), true); + + const json* p = ⌖ + for (std::size_t i = 0; i < depth; ++i) + { + p = &p->at("a"); + } + CHECK(p->size() == 2); + CHECK(p->at("x") == 1); + CHECK(p->at("y") == 2); + } +} From f290b36ad24af1d05c57af8b62111729bb374021 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 25 Sep 2026 08:23:53 +0200 Subject: [PATCH 12/17] Fix CI: use VS 2026 on windows-11-arm and use raw string literals in tests (#5575) The windows-11-arm runner image moved to windows-11-vs2026-arm64, which no longer ships Visual Studio 2022, so the msvc-arm64 job failed at configure time. Use the "Visual Studio 18 2026" generator like the msvc2026 job. clang-tidy's modernize-raw-string-literal check flagged two string literals in the nesting tests added by #5546 and #5547. Signed-off-by: Niels Lohmann --- .github/workflows/windows.yml | 4 ++-- tests/src/unit-merge_patch.cpp | 2 +- tests/src/unit-modifiers.cpp | 2 +- 3 files changed, 4 insertions(+), 4 deletions(-) diff --git a/.github/workflows/windows.yml b/.github/workflows/windows.yml index 99a8aa3ca..ea742773f 100644 --- a/.github/workflows/windows.yml +++ b/.github/workflows/windows.yml @@ -124,11 +124,11 @@ jobs: steps: - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 - name: Run CMake (Release) - run: cmake -S . -B build -G "Visual Studio 17 2022" -A ARM64 -DJSON_BuildTests=On -DCMAKE_CXX_FLAGS="/W4 /WX" + run: cmake -S . -B build -G "Visual Studio 18 2026" -A ARM64 -DJSON_BuildTests=On -DCMAKE_CXX_FLAGS="/W4 /WX" if: matrix.build_type == 'Release' shell: pwsh - name: Run CMake (Debug) - run: cmake -S . -B build -G "Visual Studio 17 2022" -A ARM64 -DJSON_BuildTests=On -DJSON_FastTests=ON -DCMAKE_CXX_FLAGS="/W4 /WX" + run: cmake -S . -B build -G "Visual Studio 18 2026" -A ARM64 -DJSON_BuildTests=On -DJSON_FastTests=ON -DCMAKE_CXX_FLAGS="/W4 /WX" if: matrix.build_type == 'Debug' shell: pwsh - name: Build diff --git a/tests/src/unit-merge_patch.cpp b/tests/src/unit-merge_patch.cpp index 8ac281ef5..c9741e85b 100644 --- a/tests/src/unit-merge_patch.cpp +++ b/tests/src/unit-merge_patch.cpp @@ -62,7 +62,7 @@ std::string nested_objects(const std::size_t depth, const int variant) } text += "\"a\":"; } - text += variant == 1 ? "{\"x\":1,\"y\":null}" : "{\"y\":2}"; + text += variant == 1 ? R"({"x":1,"y":null})" : "{\"y\":2}"; text.append(depth, '}'); return text; } diff --git a/tests/src/unit-modifiers.cpp b/tests/src/unit-modifiers.cpp index 55f9d467e..c878ec15c 100644 --- a/tests/src/unit-modifiers.cpp +++ b/tests/src/unit-modifiers.cpp @@ -48,7 +48,7 @@ std::string nested_objects(const std::size_t depth, const int variant) if (variant == 2 && i % 5 == 0) { // an object replacing a primitive, which is not merged - text += "\"s0\":{\"o\":1},"; + text += R"("s0":{"o":1},)"; } text += "\"a\":"; } From 3901b223e58671d0e8c460b530de2dfe731a269e Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 25 Sep 2026 08:28:00 +0200 Subject: [PATCH 13/17] Complete the architecture documentation page (#5570) * Complete the architecture documentation page Replace the placeholder bullets and TODOs with a description of the component pipeline (with a diagram), the source layout, the template parameters, the value storage (now in struct data), the input and output adapters, and the SAX interface. Signed-off-by: Niels Lohmann * Link sources and basic_json, document full input adapter interface Signed-off-by: Niels Lohmann * Align the default column of the template parameter table Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- docs/mkdocs/docs/home/architecture.md | 212 +++++++++++++++++++++----- 1 file changed, 177 insertions(+), 35 deletions(-) diff --git a/docs/mkdocs/docs/home/architecture.md b/docs/mkdocs/docs/home/architecture.md index aba2be580..6c0892f11 100644 --- a/docs/mkdocs/docs/home/architecture.md +++ b/docs/mkdocs/docs/home/architecture.md @@ -1,34 +1,125 @@ # Architecture -!!! info - - This page is still under construction. Its goal is to provide a high-level overview of the library's architecture. - This should help new contributors to get an idea of the used concepts and where to make changes. +This page gives a high-level overview of the library's architecture. It should help new contributors to get an idea of +the used concepts and where to make changes. ## Overview -The main structure is class [nlohmann::basic_json](../api/basic_json/index.md). +The library is built around a single class template, [`nlohmann::basic_json`](../api/basic_json/index.md). A +`basic_json` value is a node in a tree of JSON values. All other components either create such a tree from an input +(parsing), write a tree to an output (serialization), or give access to it (iterators, JSON Pointer, conversions). -- public API -- container interface -- iterators +```mermaid +flowchart LR + input[/"input
(string, stream,
iterator range, file)"/] + ia["input adapter"] + lexer["lexer"] + parser["parser"] + breader["binary_reader"] + sax["SAX interface"] + value[("basic_json
value tree")] + serializer["serializer"] + bwriter["binary_writer"] + oa["output adapter"] + output[/"output
(string, stream,
vector)"/] -## Template specializations + input --> ia + ia --> lexer --> parser --> sax + ia --> breader --> sax + sax --> value + value --> serializer --> oa + value --> bwriter --> oa + oa --> output +``` -- describe template parameters of `basic_json` -- [`json`](../api/json.md) -- [`ordered_json`](../api/ordered_json.md) via [`ordered_map`](../api/ordered_map.md) +- **JSON text** is read by an [input adapter](#input-adapters), tokenized by the lexer, and turned into SAX events by + the parser. +- **Binary formats** (BJData, BSON, CBOR, MessagePack, UBJSON) are read by an input adapter and turned into the same SAX + events by the `binary_reader`. +- A [SAX consumer](#sax-interface) receives the events. The one used by [`parse`](../api/basic_json/parse.md) builds a + `basic_json` value tree. +- The `serializer` (JSON text) or the `binary_writer` (binary formats) writes a value tree to an + [output adapter](#output-adapters). + +## Source layout + +The public headers are in [`include/nlohmann`](https://github.com/nlohmann/json/tree/develop/include/nlohmann): + +- [`json.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/json.hpp) defines class [`basic_json`](../api/basic_json/index.md). +- [`json_fwd.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/json_fwd.hpp) contains forward declarations. +- [`adl_serializer.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/adl_serializer.hpp), [`byte_container_with_subtype.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/byte_container_with_subtype.hpp), and [`ordered_map.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/ordered_map.hpp) define + [`adl_serializer`](../api/adl_serializer/index.md), + [`byte_container_with_subtype`](../api/byte_container_with_subtype/index.md), and + [`ordered_map`](../api/ordered_map.md). + +Everything else lives in [`detail/`](https://github.com/nlohmann/json/tree/develop/include/nlohmann/detail) and namespace `nlohmann::detail`, which is not part of the public API. Paths +below are relative to `include/nlohmann`. + +| Component | Location | +|-----------|----------| +| Value type enumeration | [`detail/value_t.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/value_t.hpp) | +| Input adapters | [`detail/input/input_adapters.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/input/input_adapters.hpp) | +| Lexer | [`detail/input/lexer.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/input/lexer.hpp), [`detail/input/number_parse.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/input/number_parse.hpp), [`detail/input/string_scan.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/input/string_scan.hpp) | +| Parser | [`detail/input/parser.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/input/parser.hpp) | +| SAX interface and DOM builders | [`detail/input/json_sax.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/input/json_sax.hpp) | +| Binary format readers | [`detail/input/binary_reader.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/input/binary_reader.hpp) | +| JSON serializer | [`detail/output/serializer.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/output/serializer.hpp), [`detail/conversions/to_chars.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/conversions/to_chars.hpp) | +| Binary format writers | [`detail/output/binary_writer.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/output/binary_writer.hpp) | +| Output adapters | [`detail/output/output_adapters.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/output/output_adapters.hpp) | +| Iterators | [`detail/iterators/`](https://github.com/nlohmann/json/tree/develop/include/nlohmann/detail/iterators) | +| Conversions from/to arbitrary types | [`detail/conversions/from_json.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/conversions/from_json.hpp), [`detail/conversions/to_json.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/conversions/to_json.hpp) | +| JSON Pointer | [`detail/json_pointer.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/json_pointer.hpp) | +| Exceptions | [`detail/exceptions.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/exceptions.hpp) | +| Type traits and C++ feature backports | [`detail/meta/`](https://github.com/nlohmann/json/tree/develop/include/nlohmann/detail/meta) | +| Macros | [`detail/macro_scope.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/macro_scope.hpp), [`detail/macro_unscope.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/macro_unscope.hpp), [`detail/abi_macros.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/abi_macros.hpp) | + +The single-header version [`single_include/nlohmann/json.hpp`](https://github.com/nlohmann/json/blob/develop/single_include/nlohmann/json.hpp) +is generated from these files with `make amalgamate` and must not be edited by hand. + +## Template parameters + +[`basic_json`](../api/basic_json/index.md) is parameterized by the types it uses to store values and to convert from and to other types: + +| Template parameter | Default | Used for | +|----------------------|-----------------------------|-------------------------------------------------------------------| +| `ObjectType` | `std::map` | objects, see [`object_t`](../api/basic_json/object_t.md) | +| `ArrayType` | `std::vector` | arrays, see [`array_t`](../api/basic_json/array_t.md) | +| `StringType` | `std::string` | strings and object keys, see [`string_t`](../api/basic_json/string_t.md) | +| `BooleanType` | `bool` | Booleans, see [`boolean_t`](../api/basic_json/boolean_t.md) | +| `NumberIntegerType` | `std::int64_t` | signed integers, see [`number_integer_t`](../api/basic_json/number_integer_t.md) | +| `NumberUnsignedType` | `std::uint64_t` | unsigned integers, see [`number_unsigned_t`](../api/basic_json/number_unsigned_t.md) | +| `NumberFloatType` | `double` | floating-point numbers, see [`number_float_t`](../api/basic_json/number_float_t.md) | +| `AllocatorType` | `std::allocator` | allocating objects, arrays, strings, and binary values | +| `JSONSerializer` | `adl_serializer` | conversions from/to other types, see [`adl_serializer`](../api/adl_serializer/index.md) | +| `BinaryType` | `std::vector` | binary values, see [`binary_t`](../api/basic_json/binary_t.md) | +| `CustomBaseClass` | `void` | an optional base class, see [`json_base_class_t`](../api/basic_json/json_base_class_t.md) | + +The library provides two specializations: + +- [`json`](../api/json.md) uses all default template arguments. +- [`ordered_json`](../api/ordered_json.md) uses [`ordered_map`](../api/ordered_map.md) as `ObjectType` to keep the + insertion order of object keys. + +The requirements on the template arguments are listed in +[Template Parameter Requirements](../features/types/template_parameters.md). ## Value storage -Values are stored as a tagged union of [value_t](../api/basic_json/value_t.md) and json_value. +Each [`basic_json`](../api/basic_json/index.md) value stores its content as a tagged union: an enumeration [`value_t`](../api/basic_json/value_t.md) +names the type of the value, and a union `json_value` holds the value itself. Both are members of the nested struct +`data`, which is the only data member `m_data` of `basic_json`: ```cpp -/// the type of the current element -value_t m_type = value_t::null; +struct data +{ + /// the type of the current element + value_t m_type = value_t::null; -/// the value of the current element -json_value m_value = {}; + /// the value of the current element + json_value m_value = {}; +}; + +data m_data = {}; ``` with @@ -68,42 +159,83 @@ union json_value { }; ``` -## Parsing inputs (deserialization) +Objects, arrays, strings, and binary values are allocated on the heap with `AllocatorType`, and the union only stores a +pointer to them. This keeps a `basic_json` value small: one pointer-sized union and one byte for the type. The class +maintains the invariant that the pointer matching `m_type` is never null; `assert_invariant()` checks it with +[runtime assertions](../features/assertions.md). -Input is read via **input adapters** that abstract a source with a common interface: +## Input adapters + +Input is read via **input adapters** that abstract a source. Every input adapter provides this interface: ```cpp -/// read a single character -std::char_traits::int_type get_character() noexcept; +/// the type of the characters in the input +using char_type = ...; -/// read multiple characters to a destination buffer and -/// returns the number of characters successfully read +/// read a single character; returns std::char_traits::eof() at the end of the input +typename std::char_traits::int_type get_character(); + +/// read up to count * sizeof(T) bytes into dest and return the number of bytes read +/// (used by the binary readers) template std::size_t get_elements(T* dest, std::size_t count = 1); ``` -List examples of input adapters. +The lexer detects two optional extensions at compile time. Only `iterator_input_adapter` provides them, and only for +random-access input of single-byte characters: -## SAX Interface +- `supports_seek`, `get_consumed_count()`, and `copy_consumed_range()` let the lexer reconstruct already consumed input + for error messages instead of copying every character it reads. +- `supports_bulk_scan`, `bulk_data()`, `bulk_remaining()`, and `bulk_skip()` let the lexer scan strings directly in + contiguous memory, several bytes at a time. -TODO +The function `input_adapter` picks the right adapter for the argument passed to `parse`, `accept`, `sax_parse`, or the +`from_*` functions: -## Writing outputs (serialization) +- `iterator_input_adapter` reads from an iterator range, which also covers strings, containers, and pointers. +- `wide_string_input_adapter` reads from ranges of `wchar_t`, `char16_t`, or `char32_t` and converts them to UTF-8. + It cannot be used for binary formats; its `get_elements()` throws. +- `input_stream_adapter` reads from a `std::istream`. +- `file_input_adapter` reads from a `std::FILE*`. + +## SAX interface + +The parser does not build values itself. It reports what it reads as events to a [SAX](../features/parsing/sax_interface.md) +consumer, which implements the interface [`json_sax`](../api/json_sax/index.md): `null`, `boolean`, `number_integer`, +`number_unsigned`, `number_float`, `string`, `binary`, `start_object`, `key`, `end_object`, `start_array`, `end_array`, +and `parse_error`. + +The library comes with two consumers in `detail/input/json_sax.hpp`: + +- `json_sax_dom_parser` builds a [`basic_json`](../api/basic_json/index.md) value tree. [`parse`](../api/basic_json/parse.md) uses it. +- `json_sax_dom_callback_parser` does the same, but calls a [parser callback](../features/parsing/parser_callbacks.md) + for each event, which can skip values. `parse` uses it when a callback is given. + +The `binary_reader` emits the same events for binary formats, so [`sax_parse`](../api/basic_json/sax_parse.md) works +with a user-defined consumer for JSON and for all binary formats alike. + +## Output adapters Output is written via **output adapters**: ```cpp -template void write_character(CharType c); -template void write_characters(const CharType* s, std::size_t length); ``` -List examples of output adapters. +The `serializer` (used by [`dump`](../api/basic_json/dump.md) and [`operator<<`](../api/operator_ltlt.md)) and the +`binary_writer` (used by the `to_*` functions) write to one of these adapters: + +- `output_vector_adapter` appends to a `std::vector`. +- `output_stream_adapter` writes to a `std::ostream`. +- `output_string_adapter` appends to a string. ## Value conversion +Values are converted from and to other types with the `JSONSerializer` template parameter. The default, +[`adl_serializer`](../api/adl_serializer/index.md), calls the free functions + ```cpp template void to_json(basic_json& j, const T& t); @@ -112,13 +244,23 @@ template void from_json(const basic_json& j, T& t); ``` +found by argument-dependent lookup. The library defines them for standard types in `detail/conversions`; users add them +for their own types, see [Arbitrary Type Conversions](../features/arbitrary_types.md). The +[serialization macros](../features/macros.md) generate these functions. + ## Additional features -- JSON Pointers -- Binary formats -- Custom base class -- Conversion macros +- [JSON Pointer](../features/json_pointer.md) (class `json_pointer`) addresses values inside a tree. It is also the + basis of [JSON Patch](../features/json_patch.md). +- [Binary formats](../features/binary_formats/index.md) are read by `binary_reader` and written by `binary_writer`. +- A [custom base class](../api/basic_json/json_base_class_t.md) can add members to every [`basic_json`](../api/basic_json/index.md) value. +- [Serialization macros](../features/macros.md) generate `to_json` and `from_json` functions for user-defined types. ## Details namespace -- C++ feature backports +Namespace `nlohmann::detail` contains all implementation details. It is not part of the public API and may change in any +release. Besides the components above, it contains: + +- type traits to detect the capabilities of user-defined types (`detail/meta/type_traits.hpp`), +- backports of C++14/17 features to C++11 (`detail/meta/cpp_future.hpp`), and +- helpers such as `string_concat` and `string_escape`. From aada27405dca0771eca5f8eaa1d087c4b13c0ff0 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 25 Sep 2026 08:28:16 +0200 Subject: [PATCH 14/17] Add a roadmap page to the documentation (#5571) Describe what the project will and will not do over the next year, and point to issue #3453 for the open question of a 4.0 release. Signed-off-by: Niels Lohmann --- docs/mkdocs/docs/community/index.md | 1 + docs/mkdocs/docs/community/roadmap.md | 43 +++++++++++++++++++++++++++ docs/mkdocs/mkdocs.yml | 1 + 3 files changed, 45 insertions(+) create mode 100644 docs/mkdocs/docs/community/roadmap.md diff --git a/docs/mkdocs/docs/community/index.md b/docs/mkdocs/docs/community/index.md index 50baeab25..c7d0cf079 100644 --- a/docs/mkdocs/docs/community/index.md +++ b/docs/mkdocs/docs/community/index.md @@ -5,4 +5,5 @@ - [Contribution Guidelines](contribution_guidelines.md) - guidelines how to contribute to this project - [Governance](governance.md) - the governance model of this project - [Quality Assurance](quality_assurance.md) - how the quality of this project is assured +- [Roadmap](roadmap.md) - what the project will and will not do - [Security Policy](security_policy.md) - the security policy of the project diff --git a/docs/mkdocs/docs/community/roadmap.md b/docs/mkdocs/docs/community/roadmap.md new file mode 100644 index 000000000..e8c407d3f --- /dev/null +++ b/docs/mkdocs/docs/community/roadmap.md @@ -0,0 +1,43 @@ +# Roadmap + +This page describes what the project intends to do, and what it does not intend to do, over the next year. Concrete +work items are tracked in the [GitHub milestones](https://github.com/nlohmann/json/milestones) and the +[issue tracker](https://github.com/nlohmann/json/issues). + +## What the project will do + +- **Keep the C++11 baseline.** The library will continue to compile with every + [supported C++11 compiler](https://github.com/nlohmann/json/blob/develop/README.md#supported-compilers). Features of + later standards are only used when they are guarded by the `JSON_HAS_CPP_*` macros. +- **Stay conformant to JSON.** The parser and serializer follow [RFC 8259](https://datatracker.ietf.org/doc/html/rfc8259). + Extensions such as [comments](../features/comments.md) or [trailing commas](../features/trailing_commas.md) remain + opt-in. +- **Keep the 3.x public API stable.** Releases follow [semantic versioning](https://semver.org). Changes that would + break existing code are only added behind a feature macro, so users can opt in and test their code before a next + major release. +- **Support a broad range of compilers and platforms.** The [CI](quality_assurance.md) keeps testing old and new + versions of GCC, Clang, MSVC, and other compilers on Linux, macOS, and Windows. +- **Keep the quality assurance up.** Every change keeps the test coverage at 100%, passes the static and dynamic + analysis, and is fuzz-tested by OSS-Fuzz, see [Quality assurance](quality_assurance.md). +- **Harden the library against hostile input.** Handling deeply nested values without exhausting the call stack is + ongoing work. +- **Fix bugs and security issues** reported through the issue tracker and the [security policy](security_policy.md). + +## What the project will not do + +- **Break the public API of version 3.x.** See the + [contribution guidelines](https://github.com/nlohmann/json/blob/develop/.github/CONTRIBUTING.md#break-the-public-api) + for what counts as a breaking change. +- **Require a newer C++ standard than C++11.** +- **Break JSON conformance** or enable non-standard extensions by default. +- **Add dependencies** or require a build step. The library remains header-only, and the single header + `json.hpp` remains a complete distribution. +- **Trade simplicity for speed or memory efficiency.** Performance improvements are welcome, but the library is not + meant to compete with the fastest JSON libraries, see [Design goals](../home/design_goals.md). + +## Version 4.0 + +There is no decision yet on whether or when a version 4.0 with breaking changes will be released. Proposals that need +a major version, for instance stricter type conversions, are collected in issue +[#3453](https://github.com/nlohmann/json/issues/3453). Until then, such changes are only added as opt-in behavior +behind feature macros. diff --git a/docs/mkdocs/mkdocs.yml b/docs/mkdocs/mkdocs.yml index 8f92a838f..9400222b1 100644 --- a/docs/mkdocs/mkdocs.yml +++ b/docs/mkdocs/mkdocs.yml @@ -317,6 +317,7 @@ nav: - community/contribution_guidelines.md - community/quality_assurance.md - community/governance.md + - community/roadmap.md - community/security_policy.md # Extras From 80bf54a5a2294f35b3858b602af0d9b795889b0a Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 25 Sep 2026 08:28:34 +0200 Subject: [PATCH 15/17] Add a security assurance case to the documentation (#5572) Describe the threat model, the trust boundaries, the secure-design argument, and how common weaknesses are countered, with links to the quality assurance page as evidence. Signed-off-by: Niels Lohmann --- docs/mkdocs/docs/community/assurance_case.md | 70 ++++++++++++++++++++ docs/mkdocs/docs/community/index.md | 1 + docs/mkdocs/mkdocs.yml | 1 + 3 files changed, 72 insertions(+) create mode 100644 docs/mkdocs/docs/community/assurance_case.md diff --git a/docs/mkdocs/docs/community/assurance_case.md b/docs/mkdocs/docs/community/assurance_case.md new file mode 100644 index 000000000..87d8d6f14 --- /dev/null +++ b/docs/mkdocs/docs/community/assurance_case.md @@ -0,0 +1,70 @@ +# Assurance case + +This page argues why the library meets its security requirements. It describes the threats the library faces, where the +trust boundaries lie, and how the library's design and the [quality assurance](quality_assurance.md) counter these +threats. To report a vulnerability, see the [security policy](security_policy.md). + +## Threat model + +The library parses, stores, and serializes JSON values in memory. It does not open network connections, does not open +files (it only reads from streams or `std::FILE*` handles that the caller has already opened), does not read environment +variables, and does not implement cryptography or handle credentials. + +The primary threat is therefore **untrusted input**: JSON text or binary data (BJData, BSON, CBOR, MessagePack, UBJSON) +that an attacker controls, passed to [`parse`](../api/basic_json/parse.md), [`accept`](../api/basic_json/accept.md), +[`sax_parse`](../api/basic_json/sax_parse.md), or one of the `from_*` functions such as +[`from_cbor`](../api/basic_json/from_cbor.md). Such input may try to + +- make the library read or write out of bounds (malformed lengths, truncated input, invalid UTF-8), +- trigger undefined behavior (integer overflow in sizes or numbers, invalid casts), +- exhaust memory (huge announced sizes), or +- exhaust the call stack (deeply nested arrays and objects). + +## Trust boundaries + +- **Untrusted:** all serialized input read by the parser, the SAX interface, and the binary readers. The library must + handle every possible input by either producing a value or throwing a [`parse_error`](../home/exceptions.md#parse-errors) + (or returning `false` when exceptions are disabled for the call). +- **Trusted:** the C++ code that calls the library. Calling a function with violated preconditions, for instance + accessing an array with [`operator[]`](../api/basic_json/operator%5B%5D.md) out of range, is a programming error and + not a security boundary. Such preconditions are checked with [runtime assertions](../features/assertions.md) in debug + builds; functions such as [`at`](../api/basic_json/at.md) offer checked access with exceptions. + +## Secure design + +- **Strict parsing.** The parser accepts exactly the JSON grammar of [RFC 8259](https://datatracker.ietf.org/doc/html/rfc8259). + Extensions such as [comments](../features/comments.md) and [trailing commas](../features/trailing_commas.md) must be + enabled explicitly. Invalid UTF-8 is rejected. +- **Errors are reported, not ignored.** Malformed input results in a [`parse_error`](../home/exceptions.md#parse-errors) + with the byte position of the error. Binary readers do not trust announced sizes: strings and binary values grow + only as bytes are actually read, arrays reserve at most a fixed number of elements up front, and sizes that no + container can hold are rejected. +- **Memory is owned by values.** Each `basic_json` value owns its content, and there is no manual memory management in + user code. The destructor does not recurse, so destroying a deeply nested value does not exhaust the stack. +- **Bounded recursion.** The JSON parser and the binary readers keep their state in explicit stacks instead of + recursing per nesting level. Operations that walk a value, such as [`dump`](../api/basic_json/dump.md), copying, + hashing, and [`merge_patch`](../api/basic_json/merge_patch.md), recurse only up to a fixed depth and continue with an + explicit stack below it. Some operations, such as comparison, [`diff`](../api/basic_json/diff.md), + [`flatten`](../api/basic_json/flatten.md), and the binary writers, still recurse once per nesting level; work on them + is in progress. Applications that process untrusted input can limit its nesting depth with a + [parser callback](../features/parsing/parser_callbacks.md). +- **Invariants are checked.** The class invariant (for instance, that the pointer for the stored type is never null) is + checked with runtime assertions throughout the test suite. + +## Common weaknesses + +The following table maps the relevant classes of the [Common Weakness Enumeration](https://cwe.mitre.org) to the +measures that counter them. The measures are described in detail in [Quality assurance](quality_assurance.md). + +| Weakness | Countermeasures | +|---------------------------------------------------------------------------|--------------------------------------------------------------------------------------------------| +| Out-of-bounds read/write ([CWE-125](https://cwe.mitre.org/data/definitions/125.html), [CWE-787](https://cwe.mitre.org/data/definitions/787.html)) | bounds checks on all reads from the input; AddressSanitizer and Valgrind on the test suite; OSS-Fuzz | +| Integer overflow ([CWE-190](https://cwe.mitre.org/data/definitions/190.html)) | UndefinedBehaviorSanitizer with integer overflow detection; Clang-Tidy; Cppcheck | +| Use after free, double free ([CWE-416](https://cwe.mitre.org/data/definitions/416.html), [CWE-415](https://cwe.mitre.org/data/definitions/415.html)) | ownership of all memory by values; AddressSanitizer and Valgrind; Clang Static Analyzer | +| Memory leaks ([CWE-401](https://cwe.mitre.org/data/definitions/401.html)) | Valgrind (Memcheck) on the test suite | +| Uncontrolled recursion ([CWE-674](https://cwe.mitre.org/data/definitions/674.html)) | iterative parser, binary readers, and destructor; bounded recursion in value operations; tests with deeply nested inputs | +| Uncontrolled resource consumption ([CWE-400](https://cwe.mitre.org/data/definitions/400.html)) | allocations based on announced sizes are capped; OSS-Fuzz with memory limits | +| Undefined behavior in general ([CWE-758](https://cwe.mitre.org/data/definitions/758.html)) | UndefinedBehaviorSanitizer; runtime assertions; Clang-Tidy, Cppcheck, Clang Static Analyzer, Infer | + +In addition, every line of the library is covered by the unit tests, and all parsers are fuzz-tested around the clock +by [OSS-Fuzz](https://github.com/google/oss-fuzz/tree/master/projects/json). diff --git a/docs/mkdocs/docs/community/index.md b/docs/mkdocs/docs/community/index.md index c7d0cf079..7b7f5c07a 100644 --- a/docs/mkdocs/docs/community/index.md +++ b/docs/mkdocs/docs/community/index.md @@ -7,3 +7,4 @@ - [Quality Assurance](quality_assurance.md) - how the quality of this project is assured - [Roadmap](roadmap.md) - what the project will and will not do - [Security Policy](security_policy.md) - the security policy of the project +- [Assurance Case](assurance_case.md) - why the library meets its security requirements diff --git a/docs/mkdocs/mkdocs.yml b/docs/mkdocs/mkdocs.yml index 9400222b1..ffb0fae80 100644 --- a/docs/mkdocs/mkdocs.yml +++ b/docs/mkdocs/mkdocs.yml @@ -319,6 +319,7 @@ nav: - community/governance.md - community/roadmap.md - community/security_policy.md + - community/assurance_case.md # Extras extra: From cc472af13f8a09d4caccc9df755435d06efaa3a9 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 25 Sep 2026 08:29:02 +0200 Subject: [PATCH 16/17] Check the fuzzers' UBJSON/BJData round-trip invariants in the unit tests (#5569) * Check the fuzzers' UBJSON/BJData round-trip invariants in the unit tests The strongest correctness checks for the UBJSON and BJData writers lived only in the OSS-Fuzz drivers: anything from_ubjson()/from_bjdata() returns must serialize with every option combination, parse back, and re-serialize stably. Those checks only run at OSS-Fuzz, so regressions surfaced days later as external reports - the same BJData assert pair was reported five times over three years, and #5494's harness change was followed by OSS-Fuzz 563659413 within a day. Add "UBJSON round-trip invariants" and "BJData round-trip invariants" test cases that run the drivers' checks on a fixed, deterministic corpus (tests/src/round_trip_corpus.hpp): integer and float boundaries, non-finite numbers, strings, binary values, optimized containers, deep nesting, the JData annotated-array matrix, and seeded random containers. They also check two properties the drivers do not: the first round trip preserves the value, and re-serializing reproduces the exact bytes. For BJData both exclude values containing a binary value, which is read back as an array of integers unless it was written as a Draft 3 optimized binary array; this carve-out is now documented in bjdata.md. Run against the headers before #5542, the BJData test fails, including on the shape from OSS-Fuzz 563659413. Also document how OSS-Fuzz reports are handled (reference them as "OSS-Fuzz: ", turn the reproducer into a unit test, keep drivers and unit tests in sync) in tests/fuzzing.md, and link it from the PR template and the quality assurance page. Signed-off-by: Niels Lohmann * Add the OSS-Fuzz reproducers for 474400817 and 474480402 as unit tests Following the convention added to tests/fuzzing.md, the reproducers of the two BJData fuzzer asserts tracked since January are now unit tests: - 474400817 (assert(false)): an empty object _ArraySize_ was written as the ND-array header length, which from_bjdata() could not read back. Fixed by #5455. - 474480402 (to_bjdata(j2, false, false) == vec2): a one-byte Draft 3 binary array is written in Draft 2 mode as a uint8 array and then re-serialized with the int8 marker. This is the documented exception to byte stability, not a library bug; OSS-Fuzz closed it after #5494 relaxed the harness to value stability. The test pins the exact bytes so the exception stays deliberate. The 563659413 reproducer is already a unit test (#5542). A comment also ties the existing UBJSON excessive-count test to the timeout OSS-Fuzz reported for that shape (testcase 6347769435193344). OSS-Fuzz: 474400817 OSS-Fuzz: 474480402 Signed-off-by: Niels Lohmann * Fix GCC -Weffc++ and -Wuseless-cast warnings in the round-trip corpus Initialize the atoms in the member initialization list, and drop the cast of the generator's result, which already is std::size_t on 64-bit Linux. Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- .github/PULL_REQUEST_TEMPLATE.md | 1 + .../docs/community/quality_assurance.md | 3 + .../docs/features/binary_formats/bjdata.md | 10 + tests/fuzzing.md | 23 ++ tests/src/fuzzer-parse_bjdata.cpp | 3 + tests/src/fuzzer-parse_ubjson.cpp | 3 + tests/src/round_trip_corpus.hpp | 213 ++++++++++++++++++ tests/src/unit-bjdata.cpp | 103 +++++++++ tests/src/unit-ubjson.cpp | 50 +++- 9 files changed, 408 insertions(+), 1 deletion(-) create mode 100644 tests/src/round_trip_corpus.hpp diff --git a/.github/PULL_REQUEST_TEMPLATE.md b/.github/PULL_REQUEST_TEMPLATE.md index 537095324..16d808485 100644 --- a/.github/PULL_REQUEST_TEMPLATE.md +++ b/.github/PULL_REQUEST_TEMPLATE.md @@ -2,6 +2,7 @@ - [ ] The changes are described in detail, both the what and why. - [ ] If applicable, an [existing issue](https://github.com/nlohmann/json/issues) is referenced. +- [ ] If applicable, a fixed [OSS-Fuzz](https://issues.oss-fuzz.com) issue is referenced as `OSS-Fuzz: ` (see [fuzz testing](https://github.com/nlohmann/json/blob/develop/tests/fuzzing.md#handling-oss-fuzz-reports)). - [ ] The [Code coverage](https://coveralls.io/github/nlohmann/json) remained at 100%. A test case for every new line of code. - [ ] If applicable, the [documentation](https://json.nlohmann.me) is updated. - [ ] The source code is amalgamated by running `make amalgamate`. diff --git a/docs/mkdocs/docs/community/quality_assurance.md b/docs/mkdocs/docs/community/quality_assurance.md index 4196f3532..bd35516b8 100644 --- a/docs/mkdocs/docs/community/quality_assurance.md +++ b/docs/mkdocs/docs/community/quality_assurance.md @@ -164,6 +164,9 @@ Note: Some modern features (like C++20 ranges or filesystem support) may be disa - [x] The parser is tested against extensive correctness suites for JSON compliance. - [x] In addition, the library is continuously fuzz-tested at [OSS-Fuzz](https://google.github.io/oss-fuzz/) where the library is checked against billions of inputs. +- [x] Every crash reported by OSS-Fuzz is fixed together with a unit test that reproduces it, and the fix references + the OSS-Fuzz issue. The round-trip checks of the fuzzer drivers are also part of the unit tests. See the + [fuzz testing documentation](https://github.com/nlohmann/json/blob/develop/tests/fuzzing.md#handling-oss-fuzz-reports). ## Static analysis diff --git a/docs/mkdocs/docs/features/binary_formats/bjdata.md b/docs/mkdocs/docs/features/binary_formats/bjdata.md index 4cb61053f..a0c84edaf 100644 --- a/docs/mkdocs/docs/features/binary_formats/bjdata.md +++ b/docs/mkdocs/docs/features/binary_formats/bjdata.md @@ -208,6 +208,16 @@ The library maps BJData types to JSON value types as follows: The mapping is **complete** in the sense that any BJData value can be converted to a JSON value. +!!! info "Round trips" + + A value returned by [`from_bjdata`](../../api/basic_json/from_bjdata.md) can be serialized with + [`to_bjdata`](../../api/basic_json/to_bjdata.md) using any combination of options and parsed back into an equal + value, and serializing that value again with the same options produces the same bytes. The exception is binary + values: they are only written as an optimized binary array (`[$B`) if Draft 3 is enabled and both `use_size` and + `use_type` are set. Otherwise, they are written as arrays of integers and parsed back as such (see the notes on + binary values above), and serializing such an array again may choose different, but equally valid, type markers. + The bytes can then differ, but parsing them again yields the same value. + ??? example ```cpp diff --git a/tests/fuzzing.md b/tests/fuzzing.md index cfbf4f249..b3bf90d5d 100644 --- a/tests/fuzzing.md +++ b/tests/fuzzing.md @@ -79,3 +79,26 @@ the same `fuzzers` target as above and also relies on the `FUZZER_ENGINE` variab [build script](https://github.com/google/oss-fuzz/blob/master/projects/json/build.sh) for more information. In case the build at OSS-Fuzz fails, an issue will be created automatically. + +### Handling OSS-Fuzz reports + +OSS-Fuzz files the crashes it finds in its own [issue tracker](https://issues.oss-fuzz.com), not on GitHub. So that +each report can be traced to the change that fixed it, and each fix to the report it answers, fixes follow these +conventions: + +- **Reference the OSS-Fuzz issue in the pull request**, next to any GitHub issue it closes, as `OSS-Fuzz: ` (for + example, `OSS-Fuzz: 563659413`), and in the commit message. The ID alone does not disclose the crash. If the report + was triaged into a GitHub issue, link the OSS-Fuzz issue there too. +- **Turn the reproducer into a unit test.** Download the testcase from the OSS-Fuzz report, reduce it if possible, and + add it as a regression test to the unit test of the affected format (e.g., `tests/src/unit-bjdata.cpp`), with a + comment naming the OSS-Fuzz issue. This way the input is checked by every CI run rather than only by OSS-Fuzz, and + it stays covered even if OSS-Fuzz later closes the report as not reproducible. +- **Keep the fuzzer drivers and the unit tests in sync.** The round-trip checks of the UBJSON and BJData drivers are + also run on a fixed corpus in the unit tests (see `tests/src/round_trip_corpus.hpp` and the "round-trip invariants" + test cases), so a regression shows up in CI first. When a driver's checks change, change the unit tests with them. +- **Record in the report whether the bug shipped.** OSS-Fuzz asks whether a crash was a short-lived regression or + affects a released version; answer it when the fix is merged, as it decides whether the fix needs a release note or + a security advisory (see the [security policy](../.github/SECURITY.md)). + +After the fix is merged, OSS-Fuzz re-runs the reproducer on its next build and marks the report as verified and +closed. If it does not, the fix is incomplete. diff --git a/tests/src/fuzzer-parse_bjdata.cpp b/tests/src/fuzzer-parse_bjdata.cpp index 41c51a311..d3c9e7a33 100644 --- a/tests/src/fuzzer-parse_bjdata.cpp +++ b/tests/src/fuzzer-parse_bjdata.cpp @@ -42,6 +42,9 @@ dump() serializes any non-finite double the same deterministic way (as JSON `null`, since JSON itself cannot represent NaN/Infinity), so comparing dumps is stable under exactly the same values that break operator==. +The unit tests run the same checks on a fixed corpus (see the "BJData round-trip +invariants" test case), so keep both in sync. + The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer drivers. */ diff --git a/tests/src/fuzzer-parse_ubjson.cpp b/tests/src/fuzzer-parse_ubjson.cpp index 20c20eda7..ebf775b59 100644 --- a/tests/src/fuzzer-parse_ubjson.cpp +++ b/tests/src/fuzzer-parse_ubjson.cpp @@ -21,6 +21,9 @@ array data, it performs the following steps: - j4 = from_ubjson(vec3) - assert(j1 == j4) +The unit tests run the same checks on a fixed corpus (see the "UBJSON round-trip +invariants" test case), so keep both in sync. + The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer drivers. */ diff --git a/tests/src/round_trip_corpus.hpp b/tests/src/round_trip_corpus.hpp new file mode 100644 index 000000000..41cdac3e6 --- /dev/null +++ b/tests/src/round_trip_corpus.hpp @@ -0,0 +1,213 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +#include // nan +#include // size_t +#include // int32_t, int64_t, uint32_t, uint64_t +#include // numeric_limits +#include // mt19937 +#include // string, to_string +#include // move +#include // vector + +#include + +// Values for the round-trip property tests of the UBJSON and BJData writers. +// +// The fuzzer drivers (tests/src/fuzzer-parse_ubjson.cpp and +// fuzzer-parse_bjdata.cpp) check that anything the library parses can be +// serialized, parsed back, and serialized again without loss. Those checks +// only run at OSS-Fuzz, so a regression used to surface days later as an +// external report. The unit tests run the same checks on this corpus in CI. +// +// The corpus is deterministic: std::mt19937's output sequence is fixed by +// the standard, and it is used directly rather than through a distribution +// (whose results are implementation-defined). +namespace utils +{ + +class round_trip_corpus +{ + public: + using json = nlohmann::json; + + static std::vector values() + { + round_trip_corpus corpus; + return corpus.build(); + } + + // whether a value contains a binary value, which a BJData or UBJSON round + // trip may turn into an array of integers + static bool contains_binary(const json& j) + { + if (j.is_binary()) + { + return true; + } + if (j.is_structured()) + { + for (const auto& element : j) + { + if (contains_binary(element)) + { + return true; + } + } + } + return false; + } + + private: + std::vector atoms; + // a fixed seed is the point: the corpus must be the same in every run + std::mt19937 generator{42}; // NOLINT(cert-msc32-c,cert-msc51-cpp,bugprone-random-generator-seed) + + round_trip_corpus() + : atoms + { + nullptr, true, false, + // integers at the boundaries of every UBJSON/BJData integer type + 0, 1, -1, 127, 128, 255, 256, -128, -129, + 32767, 32768, 65535, 65536, -32768, -32769, + (std::numeric_limits::min)(), (std::numeric_limits::max)(), + (std::numeric_limits::max)(), + (std::numeric_limits::min)(), (std::numeric_limits::max)(), + static_cast((std::numeric_limits::max)()) + 1u, + (std::numeric_limits::max)(), + // floating-point numbers, including non-finite ones + 0.0, -0.0, 1.5, -2.25, 3.4e38, (std::numeric_limits::max)(), + std::nan(""), std::numeric_limits::infinity(), -std::numeric_limits::infinity(), + // strings, including a non-ASCII one and one longer than 255 bytes + "", "a", "\xC3\xA4", std::string(300, 'x'), + // binary values with and without subtype + json::binary({}), json::binary({1, 2, 255}), json::binary({0x80, 0x7F}, 42), json::binary({1}, 0) + } + {} + + std::vector build() + { + std::vector result = atoms; + + // each atom inside containers, including homogeneous ones that the + // writers encode as optimized (typed) containers + result.emplace_back(json::array()); + result.emplace_back(json::object()); + for (const auto& atom : atoms) + { + result.push_back(json::array({atom})); + result.push_back(json::array({atom, atom, atom})); + result.push_back(json::array({json::array({atom})})); + result.push_back(json::object({{"key", atom}})); + } + result.push_back(json::array({1, 1.5})); + result.push_back(json::array({-1, 255})); + result.push_back(json::array({"a", "b"})); + + // deep, but well below any recursion or depth limit + json nested_array = 1; + json nested_object = 1; + for (int i = 0; i < 300; ++i) + { + nested_array = json::array({nested_array}); + nested_object = json::object({{"key", nested_object}}); + } + result.push_back(nested_array); + result.push_back(nested_object); + + add_annotated_arrays(result); + add_random_values(result); + return result; + } + + // objects in the JData annotated array format, which the BJData writer + // encodes as ND-arrays when the annotation describes a packed array, and + // as plain objects otherwise (see #5398, #5399, #5403, #5404, and #5542) + static void add_annotated_arrays(std::vector& result) + { + const std::vector types = + { + "uint8", "int8", "uint16", "int16", "uint32", "int32", "uint64", "int64", + "single", "double", "char", "byte", "bool", "unknown", 5, nullptr + }; + const std::vector sizes = + { + json::array(), {3}, {1, 3}, {3, 1}, {2, 3}, {2, 0}, {0, 2}, {2, 2, 2}, {-1, 2}, {2, 1.5}, + "3", 3, nullptr, json::binary({}) + }; + const std::vector data = + { + nullptr, 5, "s", json::object({{"a", 1}}), json::array(), + {1, 2, 3}, {1, 2, 3, 4, 5, 6}, {1, 2, 3, 4, 5, 6, 7, 8}, + {1.5, 2.5, 3.5, 4.5, 5.5, 6.5}, {300, -300, 70000, -70000, 1, 2}, + {"a", "b", "c", "d", "e", "f"}, {json::array({1, 2, 3}), json::array({4, 5, 6})} + }; + + for (const auto& type : types) + { + for (const auto& size : sizes) + { + for (const auto& d : data) + { + result.push_back({{"_ArrayType_", type}, {"_ArraySize_", size}, {"_ArrayData_", d}}); + } + } + } + + // incomplete annotations and annotations with an extra key + result.push_back({{"_ArraySize_", {2, 3}}, {"_ArrayData_", {1, 2, 3, 4, 5, 6}}}); + result.push_back({{"_ArrayType_", "uint8"}, {"_ArrayData_", {1, 2, 3, 4, 5, 6}}}); + result.push_back({{"_ArrayType_", "uint8"}, {"_ArraySize_", {2, 3}}}); + result.push_back({{"_ArrayType_", "uint8"}, {"_ArraySize_", {2, 3}}, {"_ArrayData_", {1, 2, 3, 4, 5, 6}}, {"extra", 1}}); + } + + // random containers of atoms, both homogeneous and mixed + void add_random_values(std::vector& result) + { + for (int i = 0; i < 1000; ++i) + { + result.push_back(random_value(0)); + } + } + + std::size_t random_below(std::size_t bound) + { + return generator() % bound; + } + + json random_value(int depth) + { + const auto kind = random_below(10); + if (depth > 3 || kind < 5) + { + return atoms[random_below(atoms.size())]; + } + + json result = kind < 8 ? json::array() : json::object(); + const auto count = random_below(5); + const bool homogeneous = random_below(2) == 0; + const json fixed = atoms[random_below(atoms.size())]; + for (std::size_t i = 0; i < count; ++i) + { + json element = homogeneous ? fixed : random_value(depth + 1); + if (result.is_array()) + { + result.push_back(std::move(element)); + } + else + { + result[std::to_string(i)] = std::move(element); + } + } + return result; + } +}; + +} // namespace utils diff --git a/tests/src/unit-bjdata.cpp b/tests/src/unit-bjdata.cpp index 9bba141d2..ebbbbfaf6 100644 --- a/tests/src/unit-bjdata.cpp +++ b/tests/src/unit-bjdata.cpp @@ -19,6 +19,7 @@ using nlohmann::json; #include #include #include "make_test_data_available.hpp" +#include "round_trip_corpus.hpp" #include "test_utils.hpp" namespace @@ -2867,6 +2868,21 @@ TEST_CASE("BJData") const auto out_num = json::to_bjdata(j_num); CHECK(out_num.at(0) == '{'); CHECK(json::from_bjdata(out_num) == j_num); + + // OSS-Fuzz issue 474400817: an empty object _ArraySize_ was + // written as the ND-array header length, which from_bjdata() + // could not read back + const std::vector input = + { + '[', '{', 'U', 11, '_', 'A', 'r', 'r', 'a', 'y', 'D', 'a', 't', 'a', '_', 'Z', + 'U', 11, '_', 'A', 'r', 'r', 'a', 'y', 'T', 'y', 'p', 'e', '_', 'S', 'i', 5, 'i', 'n', 't', '1', '6', + 'U', 11, '_', 'A', 'r', 'r', 'a', 'y', 'S', 'i', 'z', 'e', '_', '{', '}', '}', ']' + }; + const json j1 = json::from_bjdata(input); + CHECK(j1 == json::parse(R"([{"_ArrayType_":"int16","_ArraySize_":{},"_ArrayData_":null}])")); + json j2; + CHECK_NOTHROW(j2 = json::from_bjdata(json::to_bjdata(j1, false, false))); + CHECK(j2 == j1); } SECTION("ndarray with out-of-range _ArrayData_ elements stays as object") @@ -4273,6 +4289,93 @@ TEST_CASE("BJData use_type requires use_size") } } +TEST_CASE("BJData round-trip invariants") +{ + // This checks what the parse_bjdata_fuzzer driver checks (see + // tests/src/fuzzer-parse_bjdata.cpp), so that a regression shows up in CI + // rather than as an OSS-Fuzz report: every value from_bjdata() returns + // (j1) can be serialized with any combination of options, the result can + // be parsed back (j2), and serializing j2 again with the same options + // yields a value-equal result. + // + // Beyond the driver, this also checks that j2 equals j1 and that + // serializing j2 reproduces the exact bytes, both except for values that + // contain a binary value: a binary value is only written as a binary + // value with Draft 3's optimized binary array, and otherwise read back as + // an array of integers, for which the writer may choose different (but + // equally valid) type markers when it is serialized again (see #5494). + // + // Values are compared with dump() rather than operator==, because a NaN + // never compares equal to itself. + struct options + { + bool use_size; + bool use_type; + json::bjdata_version_t version; + }; + const std::vector all_options = + { + {false, false, json::bjdata_version_t::draft2}, + {true, false, json::bjdata_version_t::draft2}, + {true, true, json::bjdata_version_t::draft2}, + {false, false, json::bjdata_version_t::draft3}, + {true, false, json::bjdata_version_t::draft3}, + {true, true, json::bjdata_version_t::draft3}, + }; + + for (const auto& j0 : utils::round_trip_corpus::values()) + { + // turn the corpus value into a value as from_bjdata() returns it + for (const auto& initial : all_options) + { + const json j1 = json::from_bjdata(json::to_bjdata(j0, initial.use_size, initial.use_type, initial.version)); + const bool has_binary = utils::round_trip_corpus::contains_binary(j1); + + for (const auto& o : all_options) + { + INFO("j1 = " << j1.dump() << ", use_size = " << o.use_size << ", use_type = " << o.use_type + << ", draft3 = " << (o.version == json::bjdata_version_t::draft3)); + + const std::vector vec = json::to_bjdata(j1, o.use_size, o.use_type, o.version); + json j2; + // anything the library writes must be parsable by the library + REQUIRE_NOTHROW(j2 = json::from_bjdata(vec)); + const std::vector vec2 = json::to_bjdata(j2, o.use_size, o.use_type, o.version); + CHECK(json::from_bjdata(vec2).dump() == j2.dump()); + + if (!has_binary) + { + CHECK(j2.dump() == j1.dump()); + CHECK(vec2 == vec); + } + } + } + } +} + +TEST_CASE("BJData round trip of a binary value is value-stable, not byte-stable") +{ + // OSS-Fuzz issue 474480402: a Draft 3 optimized binary array is read as a + // binary value, which to_bjdata() writes in the default Draft 2 mode as a + // plain array of uint8 numbers. That is read back as an array of numbers, + // for which the writer then picks the smallest type marker, int8 ('i'), + // so re-serializing changes the bytes, but not the value. This is the + // exception described in the "Round trips" note of the BJData + // documentation, and why the fuzzer checks value stability (see #5494). + const std::vector input = {'[', '$', 'B', '#', 'U', 1, 0x20}; + const json j1 = json::from_bjdata(input); + CHECK(j1 == json::binary({0x20})); + + const std::vector vec = json::to_bjdata(j1, false, false); + CHECK(vec == std::vector({'[', 'U', 0x20, ']'})); + const json j2 = json::from_bjdata(vec); + CHECK(j2 == json::array({0x20})); + + const std::vector vec2 = json::to_bjdata(j2, false, false); + CHECK(vec2 == std::vector({'[', 'i', 0x20, ']'})); + CHECK(json::from_bjdata(vec2) == j2); +} + TEST_CASE("BJData roundtrips" * doctest::skip()) { SECTION("input from self-generated BJData files") diff --git a/tests/src/unit-ubjson.cpp b/tests/src/unit-ubjson.cpp index aafbbf5a4..c8458c44d 100644 --- a/tests/src/unit-ubjson.cpp +++ b/tests/src/unit-ubjson.cpp @@ -15,6 +15,7 @@ using nlohmann::json; #include #include #include "make_test_data_available.hpp" +#include "round_trip_corpus.hpp" #include "test_utils.hpp" namespace @@ -2265,7 +2266,9 @@ TEST_CASE("UBJSON optimized arrays of a valueless type are bounded") SECTION("an excessive count is rejected") { - // 'l' is a big-endian int32: 0x7FFFFFFF elements, about 34 GB of value + // 'l' is a big-endian int32: 0x7FFFFFFF elements, about 34 GB of value; + // OSS-Fuzz reported this shape as a parse_ubjson_fuzzer timeout + // (testcase 6347769435193344, no issue filed) for (const auto marker : {'Z', 'T', 'F' }) @@ -2817,6 +2820,51 @@ TEST_CASE("UBJSON use_type requires use_size") } } +TEST_CASE("UBJSON round-trip invariants") +{ + // This checks what the parse_ubjson_fuzzer driver checks (see + // tests/src/fuzzer-parse_ubjson.cpp), so that a regression shows up in CI + // rather than as an OSS-Fuzz report: every value from_ubjson() returns + // (j1) can be serialized with any combination of options, the result can + // be parsed back (j2), and serializing j2 again with the same options + // reproduces the exact bytes. Beyond the driver, this also checks that j2 + // equals j1. Values are compared with dump() rather than operator==, + // because a NaN never compares equal to itself. + struct options + { + bool use_size; + bool use_type; + }; + const std::vector all_options = + { + {false, false}, + {true, false}, + {true, true}, + }; + + for (const auto& j0 : utils::round_trip_corpus::values()) + { + // turn the corpus value into a value as from_ubjson() returns it; this + // has no binary values, as UBJSON writes them as arrays of integers + for (const auto& initial : all_options) + { + const json j1 = json::from_ubjson(json::to_ubjson(j0, initial.use_size, initial.use_type)); + + for (const auto& o : all_options) + { + INFO("j1 = " << j1.dump() << ", use_size = " << o.use_size << ", use_type = " << o.use_type); + + const std::vector vec = json::to_ubjson(j1, o.use_size, o.use_type); + json j2; + // anything the library writes must be parsable by the library + REQUIRE_NOTHROW(j2 = json::from_ubjson(vec)); + CHECK(j2.dump() == j1.dump()); + CHECK(json::to_ubjson(j2, o.use_size, o.use_type) == vec); + } + } + } +} + TEST_CASE("UBJSON roundtrips" * doctest::skip()) { SECTION("input from self-generated UBJSON files") From 01b53c8c15ae94e3b790ec9578a43966f0262c30 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 25 Sep 2026 08:29:36 +0200 Subject: [PATCH 17/17] Keep JSON_DIAGNOSTICS parent pointers of ordered_json members after erase() and update() (#5552) * Keep JSON_DIAGNOSTICS parent pointers of ordered_json members after erase() and update() ordered_json stores its members in a vector, and two operations moved members without restoring their parent pointers afterwards: - ordered_map::erase() re-constructs every member after the erased one in place. The basic_json move constructor leaves m_parent at nullptr, and none of the object branches of basic_json::erase() (by key, iterator, or iterator range) called set_parents(). This also affected merge_patch() with a null member and patch() with a remove operation. - update() only set the parent pointer of the inserted member. Adding a key can reallocate the vector, which copies all other members and leaves their m_parent at nullptr. The set_parents() call added for #4813 only repaired this for the nested object of a merge, not for the target. The next assert_invariant() on such an object (for instance, when copying it) aborted, and diagnostic messages lost the path prefix above the moved member. std::map-based json was not affected, because its nodes do not move. Erasing from an ordered_map object now calls set_parents(), and update() uses set_parent(), which already refreshes all members for vector-based objects. This makes the #4813 workaround redundant. Signed-off-by: Niels Lohmann * Account for JSON_DIAGNOSTIC_POSITIONS in the ordered_json parent-pointer test The merge_patch() case parses its input, so with JSON_DIAGNOSTIC_POSITIONS the exception message also carries the byte range of the parsed value. Signed-off-by: Niels Lohmann * Silence clang-tidy for the intentional copy in the ordered_json parent-pointer test Signed-off-by: Niels Lohmann * Keep parent pointers when update() merges past its descent bound The iterative path of update() only set the parent pointer of the member it inserted, like the recursive one did before. It now uses set_parent() too, so ordered_json members that move when a nested object grows keep their parents, and the set_parents() calls that patched this up after each nested merge are gone. Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- include/nlohmann/json.hpp | 48 ++++++++++----- single_include/nlohmann/json.hpp | 48 ++++++++++----- tests/src/unit-diagnostics.cpp | 100 +++++++++++++++++++++++++++++++ 3 files changed, 166 insertions(+), 30 deletions(-) diff --git a/include/nlohmann/json.hpp b/include/nlohmann/json.hpp index e40a2aa27..09dca4694 100644 --- a/include/nlohmann/json.hpp +++ b/include/nlohmann/json.hpp @@ -1243,6 +1243,27 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } + /// @brief restore the parent pointers after erasing from an object + /// ordered_json keeps its members in a vector, and erasing a member + /// re-constructs every member after it in place, which resets their + /// parent pointers + void set_parents_after_object_erase() + { +#if JSON_DIAGNOSTICS +#ifdef JSON_HEDLEY_MSVC_VERSION +#pragma warning(push ) +#pragma warning(disable : 4127) // ignore warning to replace if with if constexpr +#endif + if (detail::is_ordered_map::value) + { + set_parents(); + } +#ifdef JSON_HEDLEY_MSVC_VERSION +#pragma warning( pop ) +#endif +#endif + } + public: ////////////////////////// // JSON parser callback // @@ -2931,6 +2952,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec case value_t::object: { result.m_it.object_iterator = erase_from_object(pos.m_it.object_iterator); + set_parents_after_object_erase(); break; } @@ -3003,6 +3025,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec { result.m_it.object_iterator = m_data.m_value.object->erase(first.m_it.object_iterator, last.m_it.object_iterator); + set_parents_after_object_erase(); break; } @@ -3033,7 +3056,9 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_THROW(type_error::create(307, detail::concat("cannot use erase() with ", type_name()), this)); } - return m_data.m_value.object->erase(std::forward(key)); + const auto erased = m_data.m_value.object->erase(std::forward(key)); + set_parents_after_object_erase(); + return erased; } template < typename KeyType, detail::enable_if_t < @@ -3050,6 +3075,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec if (it != m_data.m_value.object->end()) { m_data.m_value.object->erase(it); + set_parents_after_object_erase(); return 1; } return 0; @@ -3961,16 +3987,12 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec if (it2 != m_data.m_value.object->end() && it2->second.is_object()) { it2->second.update_members(it.value().cbegin(), it.value().cend(), true, depth + 1); -#if JSON_DIAGNOSTICS - it2->second.set_parents(); -#endif continue; } } - m_data.m_value.object->operator[](it.key()) = it.value(); -#if JSON_DIAGNOSTICS - m_data.m_value.object->operator[](it.key()).m_parent = this; -#endif + // set_parent() also repairs the other members, which ordered_json + // relocates when adding a key makes its vector grow + set_parent(m_data.m_value.object->operator[](it.key()) = it.value()); } } @@ -3999,9 +4021,6 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } // a nested object is merged: continue with its parent -#if JSON_DIAGNOSTICS - target->set_parents(); -#endif target = stack.back().target; first = stack.back().position; last = stack.back().last; @@ -4023,10 +4042,9 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec continue; } } - target->m_data.m_value.object->operator[](first.key()) = first.value(); -#if JSON_DIAGNOSTICS - target->m_data.m_value.object->operator[](first.key()).m_parent = target; -#endif + // set_parent() also repairs the other members, which ordered_json + // relocates when adding a key makes its vector grow + target->set_parent(target->m_data.m_value.object->operator[](first.key()) = first.value()); ++first; } } diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 3c5821d92..9091c1198 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -25701,6 +25701,27 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } + /// @brief restore the parent pointers after erasing from an object + /// ordered_json keeps its members in a vector, and erasing a member + /// re-constructs every member after it in place, which resets their + /// parent pointers + void set_parents_after_object_erase() + { +#if JSON_DIAGNOSTICS +#ifdef JSON_HEDLEY_MSVC_VERSION +#pragma warning(push ) +#pragma warning(disable : 4127) // ignore warning to replace if with if constexpr +#endif + if (detail::is_ordered_map::value) + { + set_parents(); + } +#ifdef JSON_HEDLEY_MSVC_VERSION +#pragma warning( pop ) +#endif +#endif + } + public: ////////////////////////// // JSON parser callback // @@ -27389,6 +27410,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec case value_t::object: { result.m_it.object_iterator = erase_from_object(pos.m_it.object_iterator); + set_parents_after_object_erase(); break; } @@ -27461,6 +27483,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec { result.m_it.object_iterator = m_data.m_value.object->erase(first.m_it.object_iterator, last.m_it.object_iterator); + set_parents_after_object_erase(); break; } @@ -27491,7 +27514,9 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_THROW(type_error::create(307, detail::concat("cannot use erase() with ", type_name()), this)); } - return m_data.m_value.object->erase(std::forward(key)); + const auto erased = m_data.m_value.object->erase(std::forward(key)); + set_parents_after_object_erase(); + return erased; } template < typename KeyType, detail::enable_if_t < @@ -27508,6 +27533,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec if (it != m_data.m_value.object->end()) { m_data.m_value.object->erase(it); + set_parents_after_object_erase(); return 1; } return 0; @@ -28419,16 +28445,12 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec if (it2 != m_data.m_value.object->end() && it2->second.is_object()) { it2->second.update_members(it.value().cbegin(), it.value().cend(), true, depth + 1); -#if JSON_DIAGNOSTICS - it2->second.set_parents(); -#endif continue; } } - m_data.m_value.object->operator[](it.key()) = it.value(); -#if JSON_DIAGNOSTICS - m_data.m_value.object->operator[](it.key()).m_parent = this; -#endif + // set_parent() also repairs the other members, which ordered_json + // relocates when adding a key makes its vector grow + set_parent(m_data.m_value.object->operator[](it.key()) = it.value()); } } @@ -28457,9 +28479,6 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } // a nested object is merged: continue with its parent -#if JSON_DIAGNOSTICS - target->set_parents(); -#endif target = stack.back().target; first = stack.back().position; last = stack.back().last; @@ -28481,10 +28500,9 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec continue; } } - target->m_data.m_value.object->operator[](first.key()) = first.value(); -#if JSON_DIAGNOSTICS - target->m_data.m_value.object->operator[](first.key()).m_parent = target; -#endif + // set_parent() also repairs the other members, which ordered_json + // relocates when adding a key makes its vector grow + target->set_parent(target->m_data.m_value.object->operator[](first.key()) = first.value()); ++first; } } diff --git a/tests/src/unit-diagnostics.cpp b/tests/src/unit-diagnostics.cpp index cdb6185b3..3ae649e5b 100644 --- a/tests/src/unit-diagnostics.cpp +++ b/tests/src/unit-diagnostics.cpp @@ -361,6 +361,106 @@ TEST_CASE("Regression tests for extended diagnostics") CHECK(p == o); } } + + SECTION("Regression test - erase() and update() must keep JSON_DIAGNOSTICS parent pointers of ordered_json members") + { + // ordered_json keeps its members in a vector: erasing a member + // re-constructs all members after it in place, and adding a key may + // reallocate the vector; both reset the parent pointers of the members + // that were moved + using nlohmann::ordered_json; + + const auto check_parents = [](const ordered_json & j) + { + // const access, so operator[] cannot repair the parent pointers + CHECK_THROWS_WITH_AS(j["z"]["x"].at(0), "[json.exception.type_error.304] (/z/x) cannot use at() with number", ordered_json::type_error); + + // must not trigger assert_invariant() in a debug/assert-enabled build + ordered_json const copy = j; // NOLINT(performance-unnecessary-copy-initialization) + CHECK(copy == j); + }; + + // erase(key) + { + ordered_json j = {{"a", 1}, {"z", {{"x", 1}}}}; + CHECK(j.erase("a") == 1); + check_parents(j); + } + + // erase(iterator) + { + ordered_json j = {{"a", 1}, {"z", {{"x", 1}}}}; + j.erase(j.begin()); + check_parents(j); + } + + // erase(iterator, iterator) + { + ordered_json j = {{"a", 1}, {"b", 2}, {"z", {{"x", 1}}}}; + j.erase(j.begin(), j.find("z")); + check_parents(j); + } + + // patch() removes via erase(iterator) + { + ordered_json j = {{"a", 1}, {"z", {{"x", 1}}}}; + j.patch_inplace(ordered_json::parse(R"([{"op": "remove", "path": "/a"}])")); + check_parents(j); + } + + // update(j) + { + ordered_json j = {{"z", {{"x", 1}}}}; + j.update({{"a", 1}, {"b", 2}}); + check_parents(j); + } + + // update(j, true), the outer and the nested vector both grow + { + ordered_json j = {{"z", {{"x", 1}}}}; + j.update({{"z", {{"y", 2}}}, {"a", 1}}, true); + check_parents(j); + } + + // update(j, true) around its descent bound, where the nested vectors + // grow while the objects are merged without recursing + for (const std::size_t depth : + { + nlohmann::detail::recursion_depth_limit() - 1, nlohmann::detail::recursion_depth_limit(), nlohmann::detail::recursion_depth_limit() + 2 + }) + { + ordered_json j = {{"z", {{"x", 1}}}}; + ordered_json patch = {{"a", 1}, {"b", 2}, {"c", {{"d", 3}}}}; + for (std::size_t i = 0; i < depth; ++i) + { + j = ordered_json{{"k", 0}, {"n", std::move(j)}}; + patch = ordered_json{{"n", std::move(patch)}, {"l", 1}, {"m", 2}}; + } + j.update(patch, true); + + // must not trigger assert_invariant() on any level in a + // debug/assert-enabled build + ordered_json const copy = j; // NOLINT(performance-unnecessary-copy-initialization) + CHECK(copy == j); + } + + // merge_patch() inserts "c" and removes "d" at /a/c, then inserts "e" + // at /a, which copies /a/c + { + auto j = ordered_json::parse(R"({"a": {"c": {"d": {}}}})"); + j.merge_patch(ordered_json::parse(R"({"a": {"c": {"c": "s", "d": null}, "e": "s"}})")); + CHECK(j.dump() == R"({"a":{"c":{"c":"s"},"e":"s"}})"); + + auto const& constJ = j; +#if JSON_DIAGNOSTIC_POSITIONS + CHECK_THROWS_WITH_AS(constJ["a"]["c"]["c"].at(0), "[json.exception.type_error.304] (/a/c/c) (bytes 18-21) cannot use at() with string", ordered_json::type_error); +#else + CHECK_THROWS_WITH_AS(constJ["a"]["c"]["c"].at(0), "[json.exception.type_error.304] (/a/c/c) cannot use at() with string", ordered_json::type_error); +#endif + ordered_json const copy = j; + CHECK(copy == j); + } + } } TEST_CASE("Better diagnostics past the descent bound of update() and merge_patch()")