From 6178982b8d37f1ec3c0954baf26a69c826a79425 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Sun, 27 Sep 2026 14:21:58 +0200 Subject: [PATCH 1/5] Compare unordered objects by key below the nesting bound (#5582) * Compare unordered objects by key below the nesting bound Values nested deeper than the nesting bound are compared without the call stack, walking both objects entry by entry. Two equal objects of a type that enumerates its entries in no fixed order - std::unordered_map, say - can be walked in different orders, so they compared unequal, and a deep copy compared unequal to its original. std::unordered_map's own operator== does not depend on the order, which is what applies above the bound. Where the keys differ, equality now finds the entry by its key instead. An ordering, and ordered_map, whose operator== compares its entries in sequence, still decide by the key. Signed-off-by: Niels Lohmann * Test unordered object equality without std::unordered_map basic_json instantiates std::pair while basic_json is still incomplete. The standard does not require std::unordered_map to support that, and libstdc++ 6 to 9 as well as the EDG front ends of icpc and nvc++ reject it, which broke the build of unit-comparison on those CI jobs. The test now uses an object type derived from std::map (which, as the default object type, works everywhere) whose comparator orders keys ascending or descending as chosen at construction, and whose operator== does not depend on the order of the entries - the property of std::unordered_map the test is about. Signed-off-by: Niels Lohmann * Compare the test object type's entries with std::all_of clang-tidy (readability-use-anyofallof) asked for std::all_of instead of the loop in unordered_object_t's operator==. The entry type is spelled out, as C++11 needs typename for base_type::value_type and C++20 reports it as redundant. Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- include/nlohmann/json.hpp | 25 ++++-- single_include/nlohmann/json.hpp | 25 ++++-- tests/src/unit-comparison.cpp | 127 +++++++++++++++++++++++++++++++ 3 files changed, 167 insertions(+), 10 deletions(-) diff --git a/include/nlohmann/json.hpp b/include/nlohmann/json.hpp index d18baf599..abd461d6c 100644 --- a/include/nlohmann/json.hpp +++ b/include/nlohmann/json.hpp @@ -1492,13 +1492,28 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec compare_keys(current.lhs_object_it->first, current.rhs_object_it->first, std::integral_constant {}); - if (key_result != compare_result::equal) - { - return key_result; - } - left = &(current.lhs_object_it->second); right = &(current.rhs_object_it->second); + + if (key_result != compare_result::equal) + { + // An object type without a fixed order of its entries - + // std::unordered_map, say - may enumerate two equal + // objects differently, and its operator== does not care. + // Equality then finds the entry by its key; an ordering, + // or an object type that compares its entries in + // sequence (ordered_map), is decided by the key itself. + const auto* rhs_object = current.rhs_value->m_data.m_value.object; + const auto found = (!Ordered && !detail::is_ordered_map::value) + ? rhs_object->find(current.lhs_object_it->first) + : rhs_object->cend(); + if (found == rhs_object->cend()) + { + return key_result; + } + right = &(found->second); + } + ++current.lhs_object_it; ++current.rhs_object_it; } diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index b5938be0e..ba41273de 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -26361,13 +26361,28 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec compare_keys(current.lhs_object_it->first, current.rhs_object_it->first, std::integral_constant {}); - if (key_result != compare_result::equal) - { - return key_result; - } - left = &(current.lhs_object_it->second); right = &(current.rhs_object_it->second); + + if (key_result != compare_result::equal) + { + // An object type without a fixed order of its entries - + // std::unordered_map, say - may enumerate two equal + // objects differently, and its operator== does not care. + // Equality then finds the entry by its key; an ordering, + // or an object type that compares its entries in + // sequence (ordered_map), is decided by the key itself. + const auto* rhs_object = current.rhs_value->m_data.m_value.object; + const auto found = (!Ordered && !detail::is_ordered_map::value) + ? rhs_object->find(current.lhs_object_it->first) + : rhs_object->cend(); + if (found == rhs_object->cend()) + { + return key_result; + } + right = &(found->second); + } + ++current.lhs_object_it; ++current.rhs_object_it; } diff --git a/tests/src/unit-comparison.cpp b/tests/src/unit-comparison.cpp index 9a6606256..69c0103c9 100644 --- a/tests/src/unit-comparison.cpp +++ b/tests/src/unit-comparison.cpp @@ -15,7 +15,13 @@ #include "doctest_compatibility.h" +#include + #include +#include +#include +#include +#include #define JSON_TESTS_PRIVATE #include @@ -745,6 +751,127 @@ TEST_CASE("regression #3868 - heterogeneous comparisons compile under C++20 (P24 } #endif +namespace +{ +// orders keys ascending or descending, as chosen when a map is created +template +class directed_less +{ + public: + directed_less() = default; + + explicit directed_less(const bool descending) noexcept + : m_descending(descending) + {} + + bool operator()(const Key& lhs, const Key& rhs) const + { + return m_descending ? rhs < lhs : lhs < rhs; + } + + private: + bool m_descending = false; +}; + +// An object type that, like std::unordered_map, enumerates its entries in no +// fixed order - ascending or descending by key, depending on how the map was +// created - and whose operator== does not depend on that order. +// std::unordered_map itself cannot be used here: the standard does not +// require it to accept an incomplete mapped type such as basic_json, and +// libstdc++ 6 to 9 as well as the EDG front ends of icpc and nvc++ reject +// basic_json. std::map, the default object type, works +// with all supported compilers. +template +struct unordered_object_t : std::map, Allocator> +{ + using base_type = std::map, Allocator>; + using base_type::base_type; + + friend bool operator==(const unordered_object_t& lhs, const unordered_object_t& rhs) + { + return lhs.size() == rhs.size() && std::all_of(lhs.begin(), lhs.end(), [&rhs](const std::pair& entry) + { + const auto it = rhs.find(entry.first); + return it != rhs.end() && it->second == entry.second; + }); + } + + friend bool operator!=(const unordered_object_t& lhs, const unordered_object_t& rhs) + { + return !(lhs == rhs); + } +}; +using unordered_json = nlohmann::basic_json; + +// the entries "0" to "9", enumerated in ascending or in descending order +unordered_json make_unordered_object(const bool descending) +{ + unordered_json j = unordered_json::object_t(directed_less(descending)); + for (int i = 0; i < 10; ++i) + { + j[std::to_string(i)] = i; + } + return j; +} + +template +Json nest(Json j, const std::size_t depth) +{ + for (std::size_t i = 0; i < depth; ++i) + { + Json outer = Json::object(); + outer["x"] = std::move(j); + j = std::move(outer); + } + return j; +} +} // namespace + +TEST_CASE("equality of objects whose entries have no fixed order") +{ + // Values nested deeper than a bound are compared without the call stack, + // entry by entry. That must agree with the object type's own operator==, + // which for unordered_object_t (as for std::unordered_map) does not + // depend on the order of the entries, and for ordered_map does. + REQUIRE(make_unordered_object(true).begin().key() == "9"); + REQUIRE(make_unordered_object(false).begin().key() == "0"); + + for (const std::size_t depth : std::vector {0, 200}) + { + CAPTURE(depth); + + const unordered_json descending = nest(make_unordered_object(true), depth); + const unordered_json ascending = nest(make_unordered_object(false), depth); + CHECK(descending == ascending); + CHECK_FALSE(descending != ascending); + + // a copy is equal to its original + const unordered_json copy = descending; // NOLINT(performance-unnecessary-copy-initialization) + CHECK(copy == descending); + + // a different value, a different key, or another entry still count + unordered_json other_value = make_unordered_object(true); + other_value["5"] = 42; + CHECK_FALSE(nest(other_value, depth) == ascending); + + unordered_json other_key = make_unordered_object(true); + other_key.erase("5"); + other_key["50"] = 5; + CHECK_FALSE(nest(other_key, depth) == ascending); + + unordered_json more_entries = make_unordered_object(true); + more_entries["10"] = 10; + CHECK_FALSE(nest(more_entries, depth) == ascending); + CHECK_FALSE(ascending == nest(more_entries, depth)); + + // ordered_json compares its entries in sequence + const nlohmann::ordered_json ab = nest(nlohmann::ordered_json({{"a", 1}, {"b", 2}}), depth); + const nlohmann::ordered_json ba = nest(nlohmann::ordered_json({{"b", 2}, {"a", 1}}), depth); + CHECK_FALSE(ab == ba); + CHECK(ab != ba); + } +} + TEST_CASE("containers are compared element by element") { // Containers nested deeper than a bound are compared without the call From f7972970a4d620a9807bc63217cde94db8ff0da9 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Sun, 27 Sep 2026 14:28:16 +0200 Subject: [PATCH 2/5] Throw instead of writing MessagePack lengths beyond UINT32_MAX (#5584) * Throw instead of writing MessagePack lengths beyond UINT32_MAX MessagePack stores the length of a string, binary value, array, or object in at most 32 bits. For a larger value, to_msgpack wrote no length at all, so the output could not be read back. It now throws out_of_range.412, which BSON already uses for its 32-bit length fields. The check lives in one function, so each length is written by an if/else chain that ends in a plain else, without a condition that can never be false. It is tested with string and binary types that report a size beyond UINT32_MAX without allocating it, like the BSON tests do. Signed-off-by: Niels Lohmann * Fix the CI failures of the MessagePack length check - mark to_msgpack_length's value as used when exceptions are disabled (-Wunused-parameter, misc-unused-parameters) - put "Exception safety" before "Exceptions" in to_msgpack.md, as the documentation style check requires - create the test's string value from its type: constructing it from a beyond_uint32_string_t considers the std::filesystem::path conversion, which libstdc++ 10 reports as ambiguous for a class derived from std::string (clang 13) Signed-off-by: Niels Lohmann * Skip the MessagePack string length test for clang with libstdc++ 10 C++17 builds consider the std::filesystem::path conversion for the string type, and with clang and libstdc++ 10 that conversion is ambiguous for a class derived from std::string. Creating the value from its type did not avoid it, since any basic_json with that string type instantiates the check. The binary and ext cases are still tested there. Signed-off-by: Niels Lohmann * Keep the MessagePack string test type and its alias in one block astyle indented the alias oddly when it had an #ifdef of its own after the binary alias; declare it right after the string type, in the same block. Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- docs/mkdocs/docs/api/basic_json/to_msgpack.md | 10 +++ .../features/binary_formats/messagepack.md | 2 + docs/mkdocs/docs/home/exceptions.md | 14 +++- .../nlohmann/detail/output/binary_writer.hpp | 53 ++++++------ single_include/nlohmann/json.hpp | 53 ++++++------ tests/src/unit-msgpack.cpp | 84 ++++++++++++++++++- 6 files changed, 152 insertions(+), 64 deletions(-) diff --git a/docs/mkdocs/docs/api/basic_json/to_msgpack.md b/docs/mkdocs/docs/api/basic_json/to_msgpack.md index 66b104f52..b3bcaab7f 100644 --- a/docs/mkdocs/docs/api/basic_json/to_msgpack.md +++ b/docs/mkdocs/docs/api/basic_json/to_msgpack.md @@ -34,6 +34,15 @@ The exact mapping and its limitations are described on a [dedicated page](../../ Strong guarantee: if an exception is thrown, there are no changes in the JSON value. +## Exceptions + +- Throws [`out_of_range.412`](../../home/exceptions.md#jsonexceptionout_of_range412) if the length of a string, binary + value, array, or object exceeds 4294967295, the maximum MessagePack can store; example: + `"MessagePack length 4294967296 exceeds maximum of 4294967295"` +- Throws [`out_of_range.415`](../../home/exceptions.md#jsonexceptionout_of_range415) if the subtype of a binary value + exceeds 255, the maximum of the MessagePack ext type; example: + `"subtype 70000 is too large for the MessagePack ext type (max 255)"` + ## Complexity Linear in the size of the JSON value `j`. @@ -65,3 +74,4 @@ Linear in the size of the JSON value `j`. ## Version history - Added in version 2.0.9. +- Throws `out_of_range.412` and `out_of_range.415` since version 3.13.0. diff --git a/docs/mkdocs/docs/features/binary_formats/messagepack.md b/docs/mkdocs/docs/features/binary_formats/messagepack.md index a434909c4..0ca82c145 100644 --- a/docs/mkdocs/docs/features/binary_formats/messagepack.md +++ b/docs/mkdocs/docs/features/binary_formats/messagepack.md @@ -65,6 +65,8 @@ specification: - arrays with more than 4294967295 elements - objects with more than 4294967295 elements + Serializing such a value throws [`out_of_range.412`](../../home/exceptions.md#jsonexceptionout_of_range412). + !!! info "NaN/infinity handling" `NaN`, `Infinity`, and `-Infinity` are serialized as a MessagePack float 32 (type 0xCA, 5 bytes total), diff --git a/docs/mkdocs/docs/home/exceptions.md b/docs/mkdocs/docs/home/exceptions.md index ee76596f6..bf18baab1 100644 --- a/docs/mkdocs/docs/home/exceptions.md +++ b/docs/mkdocs/docs/home/exceptions.md @@ -932,19 +932,25 @@ A JSON Patch `add` operation cannot be applied because the target location's par ### json.exception.out_of_range.412 -BSON stores the length of documents, arrays, strings, and binary values in a signed 32-bit integer. This exception is thrown when a value is too large to be described by such a length field. +BSON stores the length of documents, arrays, strings, and binary values in a signed 32-bit integer, and MessagePack +stores the length of strings, binary values, arrays, and objects in at most an unsigned 32-bit integer. This exception +is thrown when a value is too large to be described by such a length field. -!!! failure "Example message" +!!! failure "Example messages" ``` BSON length 2147483661 exceeds maximum of 2147483647 ``` + ``` + MessagePack length 4294967296 exceeds maximum of 4294967295 + ``` !!! note - This exception was added in version 3.13.0. Before that, the length was silently truncated, and + This exception was added in version 3.13.0. Before that, the BSON length was silently truncated, and [`to_bson`](../api/basic_json/to_bson.md) produced documents with negative length prefixes that - [`from_bson`](../api/basic_json/from_bson.md) rejected. + [`from_bson`](../api/basic_json/from_bson.md) rejected; [`to_msgpack`](../api/basic_json/to_msgpack.md) wrote such + a value without any length, producing output that could not be read back. ### json.exception.out_of_range.413 diff --git a/include/nlohmann/detail/output/binary_writer.hpp b/include/nlohmann/detail/output/binary_writer.hpp index 9c41e2962..5b11a3b2f 100644 --- a/include/nlohmann/detail/output/binary_writer.hpp +++ b/include/nlohmann/detail/output/binary_writer.hpp @@ -291,6 +291,23 @@ class binary_writer } } + /*! + @brief check that @a length fits into the 32 bits that MessagePack stores + the length of a string, binary value, array, or object in + @return the length as an unsigned 32-bit integer + @throw out_of_range.412 if @a length exceeds the range of std::uint32_t + */ + static std::uint32_t to_msgpack_length(const std::size_t length, const BasicJsonType& j) + { + if (JSON_HEDLEY_UNLIKELY(!value_in_range_of(length))) + { + JSON_THROW(out_of_range::create(412, concat("MessagePack length ", std::to_string(length), " exceeds maximum of ", std::to_string((std::numeric_limits::max)())), &j)); + } + + static_cast(j); + return static_cast(length); + } + /*! @param[in] j JSON value to serialize */ @@ -430,7 +447,7 @@ class binary_writer case value_t::string: { // step 1: write control byte and the string length - const auto N = j.m_data.m_value.string->size(); + const auto N = to_msgpack_length(j.m_data.m_value.string->size(), j); if (N <= 31) { // fixstr @@ -448,17 +465,12 @@ class binary_writer oa.write_character(to_char_type(0xDA)); write_number(static_cast(N)); } - else if (N <= (std::numeric_limits::max)()) + else { // str 32 oa.write_character(to_char_type(0xDB)); write_number(static_cast(N)); } - else - { - JSON_THROW(out_of_range::create(412, concat("MessagePack size ", std::to_string(N), " exceeds maximum of ", - std::to_string((std::numeric_limits::max)())), &j)); - } // step 2: write the string oa.write_characters( @@ -470,7 +482,7 @@ class binary_writer case value_t::array: { // step 1: write control byte and the array size - const auto N = j.m_data.m_value.array->size(); + const auto N = to_msgpack_length(j.m_data.m_value.array->size(), j); if (N <= 15) { // fixarray @@ -482,17 +494,12 @@ class binary_writer oa.write_character(to_char_type(0xDC)); write_number(static_cast(N)); } - else if (N <= (std::numeric_limits::max)()) + else { // array 32 oa.write_character(to_char_type(0xDD)); write_number(static_cast(N)); } - else - { - JSON_THROW(out_of_range::create(412, concat("MessagePack size ", std::to_string(N), " exceeds maximum of ", - std::to_string((std::numeric_limits::max)())), &j)); - } // step 2: write each element for (const auto& el : *j.m_data.m_value.array) @@ -509,7 +516,7 @@ class binary_writer const bool use_ext = j.m_data.m_value.binary->has_subtype(); // step 1: write control byte and the byte string length - const auto N = j.m_data.m_value.binary->size(); + const auto N = to_msgpack_length(j.m_data.m_value.binary->size(), j); if (N <= (std::numeric_limits::max)()) { std::uint8_t output_type{}; @@ -561,7 +568,7 @@ class binary_writer oa.write_character(to_char_type(output_type)); write_number(static_cast(N)); } - else if (N <= (std::numeric_limits::max)()) + else { const std::uint8_t output_type = use_ext ? 0xC9 // ext 32 @@ -570,11 +577,6 @@ class binary_writer oa.write_character(to_char_type(output_type)); write_number(static_cast(N)); } - else - { - JSON_THROW(out_of_range::create(412, concat("MessagePack size ", std::to_string(N), " exceeds maximum of ", - std::to_string((std::numeric_limits::max)())), &j)); - } // step 1.5: if this is an ext type, write the subtype if (use_ext) @@ -598,7 +600,7 @@ class binary_writer case value_t::object: { // step 1: write control byte and the object size - const auto N = j.m_data.m_value.object->size(); + const auto N = to_msgpack_length(j.m_data.m_value.object->size(), j); if (N <= 15) { // fixmap @@ -610,17 +612,12 @@ class binary_writer oa.write_character(to_char_type(0xDE)); write_number(static_cast(N)); } - else if (N <= (std::numeric_limits::max)()) + else { // map 32 oa.write_character(to_char_type(0xDF)); write_number(static_cast(N)); } - else - { - JSON_THROW(out_of_range::create(412, concat("MessagePack size ", std::to_string(N), " exceeds maximum of ", - std::to_string((std::numeric_limits::max)())), &j)); - } // step 2: write each element for (const auto& el : *j.m_data.m_value.object) diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index ba41273de..bf1aee332 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -19756,6 +19756,23 @@ class binary_writer } } + /*! + @brief check that @a length fits into the 32 bits that MessagePack stores + the length of a string, binary value, array, or object in + @return the length as an unsigned 32-bit integer + @throw out_of_range.412 if @a length exceeds the range of std::uint32_t + */ + static std::uint32_t to_msgpack_length(const std::size_t length, const BasicJsonType& j) + { + if (JSON_HEDLEY_UNLIKELY(!value_in_range_of(length))) + { + JSON_THROW(out_of_range::create(412, concat("MessagePack length ", std::to_string(length), " exceeds maximum of ", std::to_string((std::numeric_limits::max)())), &j)); + } + + static_cast(j); + return static_cast(length); + } + /*! @param[in] j JSON value to serialize */ @@ -19895,7 +19912,7 @@ class binary_writer case value_t::string: { // step 1: write control byte and the string length - const auto N = j.m_data.m_value.string->size(); + const auto N = to_msgpack_length(j.m_data.m_value.string->size(), j); if (N <= 31) { // fixstr @@ -19913,17 +19930,12 @@ class binary_writer oa.write_character(to_char_type(0xDA)); write_number(static_cast(N)); } - else if (N <= (std::numeric_limits::max)()) + else { // str 32 oa.write_character(to_char_type(0xDB)); write_number(static_cast(N)); } - else - { - JSON_THROW(out_of_range::create(412, concat("MessagePack size ", std::to_string(N), " exceeds maximum of ", - std::to_string((std::numeric_limits::max)())), &j)); - } // step 2: write the string oa.write_characters( @@ -19935,7 +19947,7 @@ class binary_writer case value_t::array: { // step 1: write control byte and the array size - const auto N = j.m_data.m_value.array->size(); + const auto N = to_msgpack_length(j.m_data.m_value.array->size(), j); if (N <= 15) { // fixarray @@ -19947,17 +19959,12 @@ class binary_writer oa.write_character(to_char_type(0xDC)); write_number(static_cast(N)); } - else if (N <= (std::numeric_limits::max)()) + else { // array 32 oa.write_character(to_char_type(0xDD)); write_number(static_cast(N)); } - else - { - JSON_THROW(out_of_range::create(412, concat("MessagePack size ", std::to_string(N), " exceeds maximum of ", - std::to_string((std::numeric_limits::max)())), &j)); - } // step 2: write each element for (const auto& el : *j.m_data.m_value.array) @@ -19974,7 +19981,7 @@ class binary_writer const bool use_ext = j.m_data.m_value.binary->has_subtype(); // step 1: write control byte and the byte string length - const auto N = j.m_data.m_value.binary->size(); + const auto N = to_msgpack_length(j.m_data.m_value.binary->size(), j); if (N <= (std::numeric_limits::max)()) { std::uint8_t output_type{}; @@ -20026,7 +20033,7 @@ class binary_writer oa.write_character(to_char_type(output_type)); write_number(static_cast(N)); } - else if (N <= (std::numeric_limits::max)()) + else { const std::uint8_t output_type = use_ext ? 0xC9 // ext 32 @@ -20035,11 +20042,6 @@ class binary_writer oa.write_character(to_char_type(output_type)); write_number(static_cast(N)); } - else - { - JSON_THROW(out_of_range::create(412, concat("MessagePack size ", std::to_string(N), " exceeds maximum of ", - std::to_string((std::numeric_limits::max)())), &j)); - } // step 1.5: if this is an ext type, write the subtype if (use_ext) @@ -20063,7 +20065,7 @@ class binary_writer case value_t::object: { // step 1: write control byte and the object size - const auto N = j.m_data.m_value.object->size(); + const auto N = to_msgpack_length(j.m_data.m_value.object->size(), j); if (N <= 15) { // fixmap @@ -20075,17 +20077,12 @@ class binary_writer oa.write_character(to_char_type(0xDE)); write_number(static_cast(N)); } - else if (N <= (std::numeric_limits::max)()) + else { // map 32 oa.write_character(to_char_type(0xDF)); write_number(static_cast(N)); } - else - { - JSON_THROW(out_of_range::create(412, concat("MessagePack size ", std::to_string(N), " exceeds maximum of ", - std::to_string((std::numeric_limits::max)())), &j)); - } // step 2: write each element for (const auto& el : *j.m_data.m_value.object) diff --git a/tests/src/unit-msgpack.cpp b/tests/src/unit-msgpack.cpp index 0b8c6ca63..d86e6f068 100644 --- a/tests/src/unit-msgpack.cpp +++ b/tests/src/unit-msgpack.cpp @@ -14,6 +14,7 @@ using nlohmann::json; using namespace nlohmann::literals; // NOLINT(google-build-using-namespace) #endif +#include // SIZE_MAX, UINT32_MAX #include #include #include @@ -2226,7 +2227,7 @@ TEST_CASE("MessagePack Size above uint32 for array") CHECK_THROWS_WITH_AS( huge_array_json::to_msgpack(j), - "[json.exception.out_of_range.412] MessagePack size 4294967296 exceeds maximum of 4294967295", + "[json.exception.out_of_range.412] MessagePack length 4294967296 exceeds maximum of 4294967295", json::out_of_range&); array.fake_size = false; @@ -2279,7 +2280,7 @@ TEST_CASE("MessagePack Size above uint32 for object") CHECK_THROWS_WITH_AS( huge_object_json::to_msgpack(j), - "[json.exception.out_of_range.412] MessagePack size 4294967296 exceeds maximum of 4294967295", + "[json.exception.out_of_range.412] MessagePack length 4294967296 exceeds maximum of 4294967295", json::out_of_range&); object.fake_size = false; @@ -2315,7 +2316,7 @@ TEST_CASE("MessagePack Size above uint32 for string") CHECK_THROWS_WITH_AS( huge_string_json::to_msgpack(j), - "[json.exception.out_of_range.412] MessagePack size 4294967296 exceeds maximum of 4294967295", + "[json.exception.out_of_range.412] MessagePack length 4294967296 exceeds maximum of 4294967295", json::out_of_range&); } @@ -2352,7 +2353,82 @@ TEST_CASE("MessagePack Size above uint32 for binary") CHECK_THROWS_WITH_AS( huge_binary_json::to_msgpack(j), - "[json.exception.out_of_range.412] MessagePack size 4294967296 exceeds maximum of 4294967295", + "[json.exception.out_of_range.412] MessagePack length 4294967296 exceeds maximum of 4294967295", json::out_of_range&); } +namespace +{ +// types that report a size beyond UINT32_MAX without allocating that much +// memory, so the MessagePack length limit can be tested cheaply; see the +// similar types in unit-bson.cpp +std::size_t beyond_uint32_size() +{ + return static_cast((std::numeric_limits::max)()) + 1; +} + +class beyond_uint32_binary_t : public std::vector +{ + public: + using std::vector::vector; + + size_type size() const noexcept // NOLINT(readability-convert-member-functions-to-static) + { + return beyond_uint32_size(); + } +}; + +// with clang and libstdc++ 10, the std::filesystem::path conversion that +// C++17 builds consider for every string type is ambiguous for a class +// derived from std::string, so the string case is not tested there +#if !(defined(__clang__) && defined(_GLIBCXX_RELEASE) && _GLIBCXX_RELEASE < 11) + #define JSON_TEST_BEYOND_UINT32_STRING 1 +#endif + +#ifdef JSON_TEST_BEYOND_UINT32_STRING +class beyond_uint32_string_t : public std::string +{ + public: + using std::string::string; + + size_type size() const noexcept // NOLINT(readability-convert-member-functions-to-static) + { + return beyond_uint32_size(); + } +}; + +using beyond_uint32_string_json = nlohmann::basic_json < + std::map, std::vector, beyond_uint32_string_t, bool, std::int64_t, std::uint64_t, + double, std::allocator, nlohmann::adl_serializer, std::vector, void >; +#endif + +using beyond_uint32_binary_json = nlohmann::basic_json < + std::map, std::vector, std::string, bool, std::int64_t, std::uint64_t, + double, std::allocator, nlohmann::adl_serializer, beyond_uint32_binary_t, void >; +} // namespace + +TEST_CASE("MessagePack lengths beyond UINT32_MAX cannot be serialized") +{ + // MessagePack stores the length of a string, binary value, array, or + // object in at most 32 bits; a larger one used to be written without any + // length at all +#if SIZE_MAX > UINT32_MAX + { + const char* const expected = "[json.exception.out_of_range.412] MessagePack length 4294967296 exceeds maximum of 4294967295"; + + const beyond_uint32_binary_json binary = beyond_uint32_binary_json::binary(beyond_uint32_binary_t{}); + CHECK_THROWS_WITH_AS(beyond_uint32_binary_json::to_msgpack(binary), expected, beyond_uint32_binary_json::out_of_range&); + + const beyond_uint32_binary_json ext = beyond_uint32_binary_json::binary(beyond_uint32_binary_t{}, 42); + CHECK_THROWS_WITH_AS(beyond_uint32_binary_json::to_msgpack(ext), expected, beyond_uint32_binary_json::out_of_range&); + +#ifdef JSON_TEST_BEYOND_UINT32_STRING + // created from its type rather than from a beyond_uint32_string_t: + // that would consider the std::filesystem::path conversion, which + // libstdc++ 10 cannot decide for a class derived from std::string + const beyond_uint32_string_json string(beyond_uint32_string_json::value_t::string); + CHECK_THROWS_WITH_AS(beyond_uint32_string_json::to_msgpack(string), expected, beyond_uint32_string_json::out_of_range&); +#endif + } +#endif +} From 98e00d22e564de162532fb7a0da3d634af5c2bff Mon Sep 17 00:00:00 2001 From: bucketbase26 Date: Sun, 27 Sep 2026 17:58:38 +0530 Subject: [PATCH 3/5] Cut test suite runtime in binary roundtrips and integer sweeps (#5519) * Cut test suite runtime in binary roundtrips and integer sweeps The Linux CI jobs pass --no-skip, so skip() does not help there. Parse each corpus file once in the binary roundtrip loops instead of four times. Sample the 16-bit integer ranges with stride 7 (still hits every low byte) and always keep the endpoints. Also drop the 5M-node parse test to 500k, which still covers the non-recursive destructor, and move jeopardy.json into its own skipped test so the cheaper binary-format size checks actually run. See #5418. Signed-off-by: ayush-singh-0601 * Drop useless int32_t casts in the sampled integer loops ci_test_gcc compiles with -Werror=useless-cast. On that compiler int32_t is int, so static_cast of the loop bound is an error. The bounds are already int, and the sampled values do not change. Signed-off-by: ayush-singh-0601 * Revert unit-binary_formats.cpp to develop and fix comment Revert tests/src/unit-binary_formats.cpp to its develop state. The test-case split made valgrind jobs slower instead of faster, because the cheaper corpus files (canada/twitter/citm/sample) now ran under valgrind where they never did before. Fix the next_integer_sample comment: the function has no 'first' parameter, so describe what the function actually does. Signed-off-by: ayush-singh-0601 --------- Signed-off-by: ayush-singh-0601 --- tests/src/test_utils.hpp | 18 +++++++++++++++ tests/src/unit-bjdata.cpp | 32 ++++++-------------------- tests/src/unit-cbor.cpp | 42 +++++++---------------------------- tests/src/unit-large_json.cpp | 2 +- tests/src/unit-msgpack.cpp | 40 ++++++--------------------------- tests/src/unit-ubjson.cpp | 40 ++++++--------------------------- 6 files changed, 48 insertions(+), 126 deletions(-) diff --git a/tests/src/test_utils.hpp b/tests/src/test_utils.hpp index 4c81a8ef4..b2a381fb0 100644 --- a/tests/src/test_utils.hpp +++ b/tests/src/test_utils.hpp @@ -9,6 +9,7 @@ #pragma once #include // uint8_t +#include // size_t #include // ifstream, istreambuf_iterator, ios #include // vector @@ -24,6 +25,23 @@ namespace utils template inline void ignore_return_value(T&& /*unused*/) noexcept {} +// Advance i toward last (inclusive) by stride, always visiting last. +// stride 7 is coprime to 256, so every low-byte residue is still hit. +template +T next_integer_sample(T i, T last, T stride) +{ + if (i >= last) + { + return static_cast(last + 1); + } + if (stride > 0 && i > static_cast(last - stride)) + { + return last; + } + const T n = static_cast(i + stride); + return n < last ? n : last; +} + inline std::vector read_binary_file(const std::string& filename) { std::ifstream file(filename, std::ios::binary); diff --git a/tests/src/unit-bjdata.cpp b/tests/src/unit-bjdata.cpp index 03c69431b..a58507c15 100644 --- a/tests/src/unit-bjdata.cpp +++ b/tests/src/unit-bjdata.cpp @@ -418,7 +418,7 @@ TEST_CASE("BJData") SECTION("-32768..-129 (int16)") { - for (int32_t i = -32768; i <= -129; ++i) + for (int32_t i = -32768; i <= -129; i = utils::next_integer_sample(i, -129, 7)) { CAPTURE(i) @@ -578,7 +578,7 @@ TEST_CASE("BJData") SECTION("256..32767 (int16)") { - for (size_t i = 256; i <= 32767; ++i) + for (size_t i = 256; i <= 32767; i = utils::next_integer_sample(i, static_cast(32767), static_cast(7))) { CAPTURE(i) @@ -911,7 +911,7 @@ TEST_CASE("BJData") SECTION("256..32767 (int16)") { - for (size_t i = 256; i <= 32767; ++i) + for (size_t i = 256; i <= 32767; i = utils::next_integer_sample(i, static_cast(32767), static_cast(7))) { CAPTURE(i) @@ -4541,45 +4541,27 @@ TEST_CASE("BJData roundtrips" * doctest::skip()) { CAPTURE(filename) + std::ifstream f_json(filename); + const json j1 = json::parse(f_json); + auto packed = utils::read_binary_file(filename + ".bjdata"); + { INFO_WITH_TEMP(filename + ": std::vector"); - // parse JSON file - std::ifstream f_json(filename); - const json j1 = json::parse(f_json); - - // parse BJData file - auto packed = utils::read_binary_file(filename + ".bjdata"); json j2; CHECK_NOTHROW(j2 = json::from_bjdata(packed)); - - // compare parsed JSON values CHECK(j1 == j2); } { INFO_WITH_TEMP(filename + ": std::ifstream"); - // parse JSON file - std::ifstream f_json(filename); - const json j1 = json::parse(f_json); - - // parse BJData file std::ifstream f_bjdata(filename + ".bjdata", std::ios::binary); json j2; CHECK_NOTHROW(j2 = json::from_bjdata(f_bjdata)); - - // compare parsed JSON values CHECK(j1 == j2); } { INFO_WITH_TEMP(filename + ": output to output adapters"); - // parse JSON file - std::ifstream f_json(filename); - json const j1 = json::parse(f_json); - - // parse BJData file - auto packed = utils::read_binary_file(filename + ".bjdata"); - { INFO_WITH_TEMP(filename + ": output adapters: std::vector"); std::vector vec; diff --git a/tests/src/unit-cbor.cpp b/tests/src/unit-cbor.cpp index 353281d5e..77f9a10e6 100644 --- a/tests/src/unit-cbor.cpp +++ b/tests/src/unit-cbor.cpp @@ -291,7 +291,7 @@ TEST_CASE("CBOR") SECTION("-65536..-257") { - for (int32_t i = -65536; i <= -257; ++i) + for (int32_t i = -65536; i <= -257; i = utils::next_integer_sample(i, -257, 7)) { CAPTURE(i) @@ -479,7 +479,7 @@ TEST_CASE("CBOR") SECTION("256..65535") { - for (size_t i = 256; i <= 65535; ++i) + for (size_t i = 256; i <= 65535; i = utils::next_integer_sample(i, static_cast(65535), static_cast(7))) { CAPTURE(i) @@ -614,7 +614,7 @@ TEST_CASE("CBOR") SECTION("-32768..-129 (int 16)") { - for (int16_t i = -32768; i <= static_cast(-129); ++i) + for (int16_t i = -32768; i <= static_cast(-129); i = utils::next_integer_sample(i, static_cast(-129), static_cast(7))) { CAPTURE(i) @@ -719,7 +719,7 @@ TEST_CASE("CBOR") SECTION("256..65535 (two-byte uint16_t)") { - for (size_t i = 256; i <= 65535; ++i) + for (size_t i = 256; i <= 65535; i = utils::next_integer_sample(i, static_cast(65535), static_cast(7))) { CAPTURE(i) @@ -2529,60 +2529,34 @@ TEST_CASE("CBOR roundtrips" * doctest::skip()) { CAPTURE(filename) + std::ifstream f_json(filename); + const json j1 = json::parse(f_json); + const auto packed = utils::read_binary_file(filename + ".cbor"); + { INFO_WITH_TEMP(filename + ": std::vector"); - // parse JSON file - std::ifstream f_json(filename); - const json j1 = json::parse(f_json); - - // parse CBOR file - const auto packed = utils::read_binary_file(filename + ".cbor"); json j2; CHECK_NOTHROW(j2 = json::from_cbor(packed)); - - // compare parsed JSON values CHECK(j1 == j2); } { INFO_WITH_TEMP(filename + ": std::ifstream"); - // parse JSON file - std::ifstream f_json(filename); - const json j1 = json::parse(f_json); - - // parse CBOR file std::ifstream f_cbor(filename + ".cbor", std::ios::binary); json j2; CHECK_NOTHROW(j2 = json::from_cbor(f_cbor)); - - // compare parsed JSON values CHECK(j1 == j2); } { INFO_WITH_TEMP(filename + ": uint8_t* and size"); - // parse JSON file - std::ifstream f_json(filename); - const json j1 = json::parse(f_json); - - // parse CBOR file - const auto packed = utils::read_binary_file(filename + ".cbor"); json j2; CHECK_NOTHROW(j2 = json::from_cbor({packed.data(), packed.size()})); - - // compare parsed JSON values CHECK(j1 == j2); } { INFO_WITH_TEMP(filename + ": output to output adapters"); - // parse JSON file - std::ifstream f_json(filename); - json const j1 = json::parse(f_json); - - // parse CBOR file - const auto packed = utils::read_binary_file(filename + ".cbor"); - if (exclude_packed.count(filename) == 0u) { { diff --git a/tests/src/unit-large_json.cpp b/tests/src/unit-large_json.cpp index 3734966d5..8204ed9b6 100644 --- a/tests/src/unit-large_json.cpp +++ b/tests/src/unit-large_json.cpp @@ -18,7 +18,7 @@ TEST_CASE("tests on very large JSONs") { SECTION("issue #1419 - Segmentation fault (stack overflow) due to unbounded recursion") { - const auto depth = 5000000; + const auto depth = 500000; std::string s(static_cast(2 * depth), '['); std::fill(s.begin() + depth, s.end(), ']'); diff --git a/tests/src/unit-msgpack.cpp b/tests/src/unit-msgpack.cpp index d86e6f068..cfb047495 100644 --- a/tests/src/unit-msgpack.cpp +++ b/tests/src/unit-msgpack.cpp @@ -256,7 +256,7 @@ TEST_CASE("MessagePack") SECTION("256..65535 (int 16)") { - for (size_t i = 256; i <= 65535; ++i) + for (size_t i = 256; i <= 65535; i = utils::next_integer_sample(i, static_cast(65535), static_cast(7))) { CAPTURE(i) @@ -441,7 +441,7 @@ TEST_CASE("MessagePack") SECTION("-32768..-129 (int 16)") { - for (int16_t i = -32768; i <= static_cast(-129); ++i) + for (int16_t i = -32768; i <= static_cast(-129); i = utils::next_integer_sample(i, static_cast(-129), static_cast(7))) { CAPTURE(i) @@ -647,7 +647,7 @@ TEST_CASE("MessagePack") SECTION("256..65535 (uint 16)") { - for (size_t i = 256; i <= 65535; ++i) + for (size_t i = 256; i <= 65535; i = utils::next_integer_sample(i, static_cast(65535), static_cast(7))) { CAPTURE(i) @@ -2043,60 +2043,34 @@ TEST_CASE("MessagePack roundtrips" * doctest::skip()) { CAPTURE(filename) + std::ifstream f_json(filename); + const json j1 = json::parse(f_json); + auto packed = utils::read_binary_file(filename + ".msgpack"); + { INFO_WITH_TEMP(filename + ": std::vector"); - // parse JSON file - std::ifstream f_json(filename); - const json j1 = json::parse(f_json); - - // parse MessagePack file - auto packed = utils::read_binary_file(filename + ".msgpack"); json j2; CHECK_NOTHROW(j2 = json::from_msgpack(packed)); - - // compare parsed JSON values CHECK(j1 == j2); } { INFO_WITH_TEMP(filename + ": std::ifstream"); - // parse JSON file - std::ifstream f_json(filename); - const json j1 = json::parse(f_json); - - // parse MessagePack file std::ifstream f_msgpack(filename + ".msgpack", std::ios::binary); json j2; CHECK_NOTHROW(j2 = json::from_msgpack(f_msgpack)); - - // compare parsed JSON values CHECK(j1 == j2); } { INFO_WITH_TEMP(filename + ": uint8_t* and size"); - // parse JSON file - std::ifstream f_json(filename); - const json j1 = json::parse(f_json); - - // parse MessagePack file - auto packed = utils::read_binary_file(filename + ".msgpack"); json j2; CHECK_NOTHROW(j2 = json::from_msgpack({packed.data(), packed.size()})); - - // compare parsed JSON values CHECK(j1 == j2); } { INFO_WITH_TEMP(filename + ": output to output adapters"); - // parse JSON file - std::ifstream f_json(filename); - json const j1 = json::parse(f_json); - - // parse MessagePack file - auto packed = utils::read_binary_file(filename + ".msgpack"); - if (exclude_packed.count(filename) == 0u) { { diff --git a/tests/src/unit-ubjson.cpp b/tests/src/unit-ubjson.cpp index c8a300e72..c315fac94 100644 --- a/tests/src/unit-ubjson.cpp +++ b/tests/src/unit-ubjson.cpp @@ -265,7 +265,7 @@ TEST_CASE("UBJSON") SECTION("-32768..-129 (int16)") { - for (int32_t i = -32768; i <= -129; ++i) + for (int32_t i = -32768; i <= -129; i = utils::next_integer_sample(i, -129, 7)) { CAPTURE(i) @@ -425,7 +425,7 @@ TEST_CASE("UBJSON") SECTION("256..32767 (int16)") { - for (size_t i = 256; i <= 32767; ++i) + for (size_t i = 256; i <= 32767; i = utils::next_integer_sample(i, static_cast(32767), static_cast(7))) { CAPTURE(i) @@ -631,7 +631,7 @@ TEST_CASE("UBJSON") SECTION("256..32767 (int16)") { - for (size_t i = 256; i <= 32767; ++i) + for (size_t i = 256; i <= 32767; i = utils::next_integer_sample(i, static_cast(32767), static_cast(7))) { CAPTURE(i) @@ -2980,60 +2980,34 @@ TEST_CASE("UBJSON roundtrips" * doctest::skip()) { CAPTURE(filename) + std::ifstream f_json(filename); + json const j1 = json::parse(f_json); + auto const packed = utils::read_binary_file(filename + ".ubjson"); + { INFO_WITH_TEMP(filename + ": std::vector"); - // parse JSON file - std::ifstream f_json(filename); - json const j1 = json::parse(f_json); - - // parse UBJSON file - auto const packed = utils::read_binary_file(filename + ".ubjson"); json j2; CHECK_NOTHROW(j2 = json::from_ubjson(packed)); - - // compare parsed JSON values CHECK(j1 == j2); } { INFO_WITH_TEMP(filename + ": std::ifstream"); - // parse JSON file - std::ifstream f_json(filename); - json const j1 = json::parse(f_json); - - // parse UBJSON file std::ifstream f_ubjson(filename + ".ubjson", std::ios::binary); json j2; CHECK_NOTHROW(j2 = json::from_ubjson(f_ubjson)); - - // compare parsed JSON values CHECK(j1 == j2); } { INFO_WITH_TEMP(filename + ": uint8_t* and size"); - // parse JSON file - std::ifstream f_json(filename); - const json j1 = json::parse(f_json); - - // parse UBJSON file - auto const packed = utils::read_binary_file(filename + ".ubjson"); json j2; CHECK_NOTHROW(j2 = json::from_ubjson({packed.data(), packed.size()})); - - // compare parsed JSON values CHECK(j1 == j2); } { INFO_WITH_TEMP(filename + ": output to output adapters"); - // parse JSON file - std::ifstream f_json(filename); - json const j1 = json::parse(f_json); - - // parse UBJSON file - auto const packed = utils::read_binary_file(filename + ".ubjson"); - { INFO_WITH_TEMP(filename + ": output adapters: std::vector"); std::vector vec; From 6bd106893a2b56296711f71fa2a80ecf9e9798cd Mon Sep 17 00:00:00 2001 From: Kartikey Negi <65110918+ReturnKartikey@users.noreply.github.com> Date: Sun, 27 Sep 2026 17:58:55 +0530 Subject: [PATCH 4/5] Fix CBOR tag handling in cbor_tag_handler_t::store for non-binary items (#5559) When using cbor_tag_handler_t::store, tags 0xD8-0xDB previously assumed that the tagged item was a byte string, unconditionally attempting to parse binary data and failing on valid CBOR documents containing tags applied to integers, strings, arrays, or objects (such as self-describe tag 55799). Check whether the tagged data item is a byte string (0x40-0x5B or 0x5F). If it is a byte string, store the subtype on the binary value as before. Otherwise, iteratively process the tagged value in the driver loop using item_read so that chained tags do not consume native stack space. Part of #5316. Signed-off-by: ReturnKartikey --- .../docs/api/basic_json/cbor_tag_handler_t.md | 2 +- .../docs/features/binary_formats/cbor.md | 2 +- .../nlohmann/detail/input/binary_reader.hpp | 25 ++++-- single_include/nlohmann/json.hpp | 25 ++++-- tests/src/unit-cbor.cpp | 85 +++++++++++++++++++ 5 files changed, 127 insertions(+), 12 deletions(-) diff --git a/docs/mkdocs/docs/api/basic_json/cbor_tag_handler_t.md b/docs/mkdocs/docs/api/basic_json/cbor_tag_handler_t.md index e19c3edd9..cea009e4e 100644 --- a/docs/mkdocs/docs/api/basic_json/cbor_tag_handler_t.md +++ b/docs/mkdocs/docs/api/basic_json/cbor_tag_handler_t.md @@ -18,7 +18,7 @@ ignore : ignore tags store -: store tagged values as binary container with subtype (for bytes 0xd8..0xdb) +: store tagged byte strings (for bytes 0xd8..0xdb) as binary values with the tag as subtype; other tagged values are read as if the tag were ignored. If several tags precede a byte string, only the innermost one is stored. ## Examples diff --git a/docs/mkdocs/docs/features/binary_formats/cbor.md b/docs/mkdocs/docs/features/binary_formats/cbor.md index e4c257e27..8e6acf0fb 100644 --- a/docs/mkdocs/docs/features/binary_formats/cbor.md +++ b/docs/mkdocs/docs/features/binary_formats/cbor.md @@ -188,7 +188,7 @@ The library maps CBOR types to JSON value types as follows: !!! warning "Tagged items" - Tagged items (0xC0..0xDB) will throw a parse error by default. They can be ignored by passing `cbor_tag_handler_t::ignore` to function `from_cbor`, in which case the tag is skipped and the enclosed data item is parsed on its own. They can be stored by passing `cbor_tag_handler_t::store` to function `from_cbor`. Note that no tag is ever interpreted: for instance, a text string tagged with tag 0 (date/time) stays a string. + Tagged items (0xC0..0xDB) will throw a parse error by default. They can be ignored by passing `cbor_tag_handler_t::ignore` to function `from_cbor`, in which case the tag is skipped and the enclosed data item is parsed on its own. Passing `cbor_tag_handler_t::store` to function `from_cbor` stores tagged byte strings (for bytes 0xd8..0xdb) as binary values with the tag as subtype; other tagged values are read as if the tag were ignored. If several tags precede a byte string, only the innermost one is stored. Note that no tag is ever interpreted: for instance, a text string tagged with tag 0 (date/time) stays a string. ??? example diff --git a/include/nlohmann/detail/input/binary_reader.hpp b/include/nlohmann/detail/input/binary_reader.hpp index 4132c03ca..b6c219fc1 100644 --- a/include/nlohmann/detail/input/binary_reader.hpp +++ b/include/nlohmann/detail/input/binary_reader.hpp @@ -44,7 +44,7 @@ enum class cbor_tag_handler_t { error, ///< throw a parse_error exception in case of a tag ignore, ///< ignore tags - store ///< store tags as binary type + store ///< store tagged byte strings (for bytes 0xd8..0xdb) as binary values with the tag as subtype; other tagged values are read as if the tag were ignored }; /*! @@ -592,14 +592,18 @@ class binary_reader input (true) or whether the last read character should be considered instead (false) @param[in] tag_handler how CBOR tags should be treated + @param[out] tag_pending whether a tag was parsed and its value follows + @param[out] item_read whether the tagged value's initial byte is already in current @return whether a valid CBOR value was passed to the SAX parser */ bool parse_cbor_value(const bool get_char, const cbor_tag_handler_t tag_handler, - bool& tag_pending) + bool& tag_pending, + bool& item_read) { tag_pending = false; + item_read = false; switch (get_char ? get() : current) { @@ -1021,7 +1025,17 @@ class binary_reader } } get(); - return get_cbor_binary(b) && sax->binary(b); + // a byte string (the heads accepted by get_cbor_binary) keeps the tag as subtype + if ((current >= 0x40 && current <= 0x5B) || current == 0x5F) + { + return get_cbor_binary(b) && sax->binary(b); + } + + // not a byte string: the tagged value, whose first byte + // was just read, is read by the caller like for ignore + tag_pending = true; + item_read = true; + return true; } default: // LCOV_EXCL_LINE @@ -1503,13 +1517,14 @@ class binary_reader // a tag is not a value of its own: read on until the tagged value bool tag_pending = false; + bool item_read = false; do { - if (JSON_HEDLEY_UNLIKELY(!parse_cbor_value(fetch, tag_handler, tag_pending))) + if (JSON_HEDLEY_UNLIKELY(!parse_cbor_value(fetch, tag_handler, tag_pending, item_read))) { return false; } - fetch = true; + fetch = !item_read; } while (tag_pending); diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index bf1aee332..81ddfb1c4 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -12741,7 +12741,7 @@ enum class cbor_tag_handler_t { error, ///< throw a parse_error exception in case of a tag ignore, ///< ignore tags - store ///< store tags as binary type + store ///< store tagged byte strings (for bytes 0xd8..0xdb) as binary values with the tag as subtype; other tagged values are read as if the tag were ignored }; /*! @@ -13289,14 +13289,18 @@ class binary_reader input (true) or whether the last read character should be considered instead (false) @param[in] tag_handler how CBOR tags should be treated + @param[out] tag_pending whether a tag was parsed and its value follows + @param[out] item_read whether the tagged value's initial byte is already in current @return whether a valid CBOR value was passed to the SAX parser */ bool parse_cbor_value(const bool get_char, const cbor_tag_handler_t tag_handler, - bool& tag_pending) + bool& tag_pending, + bool& item_read) { tag_pending = false; + item_read = false; switch (get_char ? get() : current) { @@ -13718,7 +13722,17 @@ class binary_reader } } get(); - return get_cbor_binary(b) && sax->binary(b); + // a byte string (the heads accepted by get_cbor_binary) keeps the tag as subtype + if ((current >= 0x40 && current <= 0x5B) || current == 0x5F) + { + return get_cbor_binary(b) && sax->binary(b); + } + + // not a byte string: the tagged value, whose first byte + // was just read, is read by the caller like for ignore + tag_pending = true; + item_read = true; + return true; } default: // LCOV_EXCL_LINE @@ -14200,13 +14214,14 @@ class binary_reader // a tag is not a value of its own: read on until the tagged value bool tag_pending = false; + bool item_read = false; do { - if (JSON_HEDLEY_UNLIKELY(!parse_cbor_value(fetch, tag_handler, tag_pending))) + if (JSON_HEDLEY_UNLIKELY(!parse_cbor_value(fetch, tag_handler, tag_pending, item_read))) { return false; } - fetch = true; + fetch = !item_read; } while (tag_pending); diff --git a/tests/src/unit-cbor.cpp b/tests/src/unit-cbor.cpp index 77f9a10e6..6bd792f8a 100644 --- a/tests/src/unit-cbor.cpp +++ b/tests/src/unit-cbor.cpp @@ -2124,6 +2124,20 @@ TEST_CASE("CBOR nesting does not consume the call stack") CHECK(json::from_cbor(input, true, false, json::cbor_tag_handler_t::ignore).is_discarded()); } + SECTION("stored tags") + { + // a tag over something other than a byte string is read like for + // ignore, so a chain of them must not recurse either (#5316) + std::vector input; + for (std::size_t i = 0; i < 500000; ++i) + { + input.push_back(0xD8); + input.push_back(0x18); + } + input.push_back(0x01); + CHECK(json::from_cbor(input, true, true, json::cbor_tag_handler_t::store) == 1); + } + SECTION("a well-formed deep value is read through the SAX interface") { std::vector input(200000, 0x9F); @@ -3054,6 +3068,77 @@ TEST_CASE("Tagged values") CHECK_THROWS_AS(_ = json::from_cbor(v_tagged, true, true, json::cbor_tag_handler_t::error), json::parse_error); CHECK_THROWS_AS(_ = json::from_cbor(v_tagged, true, true, json::cbor_tag_handler_t::ignore), json::parse_error); } + + SECTION("issue #5316 - cbor_tag_handler_t::store on non-binary tagged items") + { + // 55799({"a": 1}) -- CBOR self-describe magic followed by a map + const std::vector v_map{0xD9, 0xD9, 0xF7, 0xA1, 0x61, 0x61, 0x01}; + CHECK(json::from_cbor(v_map, true, true, json::cbor_tag_handler_t::ignore) == json({{"a", 1}})); + CHECK(json::from_cbor(v_map, true, true, json::cbor_tag_handler_t::store) == json({{"a", 1}})); + + // Tag 24 over unsigned integer 5 + const std::vector v_int{0xD8, 0x18, 0x05}; + CHECK(json::from_cbor(v_int, true, true, json::cbor_tag_handler_t::ignore) == 5); + CHECK(json::from_cbor(v_int, true, true, json::cbor_tag_handler_t::store) == 5); + + // Tag 24 over text string "foo" + const std::vector v_str{0xD8, 0x18, 0x63, 'f', 'o', 'o'}; + CHECK(json::from_cbor(v_str, true, true, json::cbor_tag_handler_t::ignore) == "foo"); + CHECK(json::from_cbor(v_str, true, true, json::cbor_tag_handler_t::store) == "foo"); + + // Tag 24 over array [1, 2] + const std::vector v_arr{0xD8, 0x18, 0x82, 0x01, 0x02}; + CHECK(json::from_cbor(v_arr, true, true, json::cbor_tag_handler_t::ignore) == json({1, 2})); + CHECK(json::from_cbor(v_arr, true, true, json::cbor_tag_handler_t::store) == json({1, 2})); + + // Tag 24 over boolean true + const std::vector v_bool{0xD8, 0x18, 0xF5}; + CHECK(json::from_cbor(v_bool, true, true, json::cbor_tag_handler_t::ignore) == true); + CHECK(json::from_cbor(v_bool, true, true, json::cbor_tag_handler_t::store) == true); + + // Tag 24 over null + const std::vector v_null{0xD8, 0x18, 0xF6}; + CHECK(json::from_cbor(v_null, true, true, json::cbor_tag_handler_t::ignore) == nullptr); + CHECK(json::from_cbor(v_null, true, true, json::cbor_tag_handler_t::store) == nullptr); + + // Nested tags: tag 55799 over tag 24 over integer 42 + const std::vector v_nested{0xD9, 0xD9, 0xF7, 0xD8, 0x18, 0x18, 0x2A}; + CHECK(json::from_cbor(v_nested, true, true, json::cbor_tag_handler_t::ignore) == 42); + CHECK(json::from_cbor(v_nested, true, true, json::cbor_tag_handler_t::store) == 42); + + // Tag 24 over byte string continues to store subtype as before + const std::vector v_bin{0xD8, 0x18, 0x42, 0xCA, 0xFE}; + auto j_bin_store = json::from_cbor(v_bin, true, true, json::cbor_tag_handler_t::store); + CHECK(j_bin_store.is_binary()); + CHECK(j_bin_store.get_binary().has_subtype()); + CHECK(j_bin_store.get_binary().subtype() == 24); + CHECK(j_bin_store.get_binary() == json::binary({0xCA, 0xFE}, 24).get_binary()); + + // Tagged values inside a container under store: [24(1), 25(h'0001')] + const std::vector v_container{0x82, 0xD8, 0x18, 0x01, 0xD8, 0x19, 0x42, 0x00, 0x01}; + auto j_container_store = json::from_cbor(v_container, true, true, json::cbor_tag_handler_t::store); + CHECK(j_container_store.is_array()); + CHECK(j_container_store.size() == 2); + CHECK(j_container_store[0] == 1); + CHECK(j_container_store[1].is_binary()); + CHECK(j_container_store[1].get_binary().has_subtype()); + CHECK(j_container_store[1].get_binary().subtype() == 25); + CHECK(j_container_store[1].get_binary() == json::binary({0x00, 0x01}, 25).get_binary()); + + // Tagged values as object values under store: {"a": 55799(1), "b": 24(h'01')} + const std::vector v_object{0xA2, 0x61, 'a', 0xD9, 0xD9, 0xF7, 0x01, 0x61, 'b', 0xD8, 0x18, 0x41, 0x01}; + CHECK(json::from_cbor(v_object, true, true, json::cbor_tag_handler_t::store) == json({{"a", 1}, {"b", json::binary({0x01}, 24)}})); + + // two tags in a row before a byte string: the inner tag is stored + // (this uses item_read and then the byte-string path) + const std::vector v_nested_byte_string{0xD8, 0x18, 0xD8, 0x19, 0x42, 0x00, 0x01}; + CHECK(json::from_cbor(v_nested_byte_string, true, true, json::cbor_tag_handler_t::store) == json::binary({0x00, 0x01}, 25)); + + // errors after a stored tag are now the same as with ignore + json _; + CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector {0xD8, 0x18}, true, true, json::cbor_tag_handler_t::store), "[json.exception.parse_error.110] parse error at byte 3: syntax error while parsing CBOR value: unexpected end of input", json::parse_error&); + CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector {0xD8, 0x18, 0x1C}, true, true, json::cbor_tag_handler_t::store), "[json.exception.parse_error.112] parse error at byte 3: syntax error while parsing CBOR value: invalid byte: 0x1C", json::parse_error&); + } } SECTION("negative integer overflow") From f682cd2ef1677e1218d714abb177fa2a93eadccf Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Sun, 27 Sep 2026 15:58:59 +0200 Subject: [PATCH 5/5] Skip the #5515 MessagePack size tests on 32-bit platforms (#5590) The tests fake a container size of UINT32_MAX + 1, which does not fit into a 32-bit std::size_t: MSVC rejects the truncation (C4305/C4309 with /WX), and clang-cl wraps the size to 0 so nothing throws. Guard them with SIZE_MAX > UINT32_MAX like the tests from #5584. Signed-off-by: Niels Lohmann --- tests/src/unit-msgpack.cpp | 3 +++ 1 file changed, 3 insertions(+) diff --git a/tests/src/unit-msgpack.cpp b/tests/src/unit-msgpack.cpp index cfb047495..e7616b27f 100644 --- a/tests/src/unit-msgpack.cpp +++ b/tests/src/unit-msgpack.cpp @@ -2164,6 +2164,8 @@ TEST_CASE("MessagePack with std::byte") } #endif +// the fake sizes below do not fit into a 32-bit std::size_t +#if SIZE_MAX > UINT32_MAX template> struct huge_array : std::vector { @@ -2330,6 +2332,7 @@ TEST_CASE("MessagePack Size above uint32 for binary") "[json.exception.out_of_range.412] MessagePack length 4294967296 exceeds maximum of 4294967295", json::out_of_range&); } +#endif namespace {