From 43fe8928e1a7a38569c2497e26fbd067d6d40810 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Wed, 7 Oct 2026 19:18:20 +0200 Subject: [PATCH] Stop binary writers overflowing the stack on deep values (#5781) * Stop binary writers overflowing the stack on deep values to_cbor, to_msgpack, and to_ubjson recurse once per nesting level. The parser is iterative, so a value the library accepts can crash on the way back out. Keep the existing recursive path for the first 128 levels and finish anything deeper on a heap stack. Output is unchanged. BSON is left alone because its extra size walk is a separate change. Rebased onto the value-type output sink. The heap frames now initialize every member, which is what -Weffc++ was rejecting. See #5392. Signed-off-by: ayush-singh-0601 (cherry picked from commit cf65ac438f5d43aa9a5cbf3f25c0d20fac51d4ae) Signed-off-by: Niels Lohmann * Redesign the iterative binary writers around a shared recursion depth limit Address the open review on the non-recursive CBOR/MessagePack/UBJSON/BJData writers (#5518): - Delete the CBOR array/object prefix helpers; both the recursive and iterative paths call write_cbor_head(), which already existed on develop. - MessagePack: share one write_msgpack_array_prefix()/write_msgpack_object_prefix() helper per container kind between the recursive and iterative paths, both going through to_msgpack_length() so an over-long container throws out_of_range.412 identically either way. - Reuse detail::recursion_depth_limit() instead of a separate constant, the same bound serializer::dump() and write_bson_document() already use. - Redesign the frames after bson_frame/dump_frame: only a container with elements is ever pushed, its header is written at the point it is pushed, and the iterator is set in the frame's constructor instead of a default-then-assign two-step with a since-removed "started" flag. The UBJSON frame keeps only the value pointer, the per-element prefix_required flag, and the iterator; write_closer and is_object are no longer stored, since the former is always !use_count (use_count is constant for the whole document) and the latter follows from value->is_object(). - Factor the BJData ND-array shape check into is_bjdata_ndarray(), used by both the recursive object case and the iterative pushing logic. - Give the frame classes the GCC -Weffc++ treatment already used for diff_frame: a noexcept converting constructor plus the five special members defaulted with no explicit noexcept. - Fix two @ref self-references in write_cbor/write_msgpack/write_ubjson's own doc comments to point at the public to_cbor/to_msgpack/to_ubjson/to_bjdata API instead. - The iterative object-key write for CBOR/MessagePack now runs the same strict-mode check_utf8() against the parent object as diagnostics context that the recursive path already ran, so the two paths raise identical diagnostics across the switch-over. - Rewrite the tests: round trips instead of a bare size check, byte-exact comparisons against the recursive output at depths around the bound, a deep object and a BJData ND-array past the bound, a deep discarded value (type_error.321), and the OSS-Fuzz 566583014 CBOR/MessagePack regression. BSON is unaffected by this change; it already walks its documents iteratively and is covered separately by #5553. Co-authored-by: ayush-singh-0601 <179524189+ayush-singh-0601@users.noreply.github.com> Signed-off-by: Niels Lohmann --------- Signed-off-by: ayush-singh-0601 Signed-off-by: Niels Lohmann Co-authored-by: ayush-singh-0601 Co-authored-by: ayush-singh-0601 <179524189+ayush-singh-0601@users.noreply.github.com> --- .../nlohmann/detail/output/binary_writer.hpp | 696 +++++++++++++---- single_include/nlohmann/json.hpp | 697 ++++++++++++++---- tests/src/unit-large_json.cpp | 197 +++++ 3 files changed, 1338 insertions(+), 252 deletions(-) diff --git a/include/nlohmann/detail/output/binary_writer.hpp b/include/nlohmann/detail/output/binary_writer.hpp index d5290d2c7..1cb64f08d 100644 --- a/include/nlohmann/detail/output/binary_writer.hpp +++ b/include/nlohmann/detail/output/binary_writer.hpp @@ -28,6 +28,7 @@ #include #include #include +#include #include #include @@ -156,13 +157,30 @@ class binary_writer } /*! - @param[in] j JSON value to serialize + @param[in] j JSON value to serialize + @param[in] depth nesting level of @a j, counted from the top-level value + passed to @ref basic_json::to_cbor @throw type_error.316 if a string value or an object key is not valid UTF-8 @throw type_error.321 if @a j or a value nested in it is discarded + + Serializing a container descends into its elements, so a value nested deeply + enough used to exhaust the call stack and terminate the process with no + exception to catch. The descent is bounded here: once @ref recursion_depth_limit + levels have been entered, @ref write_cbor_iterative writes out what is left + without the call stack. A value nested less deeply than that - all but a + vanishing minority - is written by exactly the code that always wrote it. + + @sa https://github.com/nlohmann/json/issues/5392 */ - void write_cbor(const BasicJsonType& j) + void write_cbor(const BasicJsonType& j, const std::size_t depth = 0) { + if (JSON_HEDLEY_UNLIKELY(depth >= recursion_depth_limit()) && (j.is_array() || j.is_object())) + { + write_cbor_iterative(j); + return; + } + switch (j.type()) { case value_t::null: @@ -244,10 +262,9 @@ class binary_writer // step 1: write control byte and the array size write_cbor_head(0x80, j.m_data.m_value.array->size()); - // step 2: write each element for (const auto& el : *j.m_data.m_value.array) { - write_cbor(el); + write_cbor(el, depth + 1); } break; } @@ -302,7 +319,6 @@ class binary_writer // step 1: write control byte and the object size write_cbor_head(0xA0, j.m_data.m_value.object->size()); - // step 2: write each element for (const auto& el : *j.m_data.m_value.object) { // el.first is checked here, against the object as @@ -317,7 +333,7 @@ class binary_writer check_utf8(el.first, j); } write_cbor(el.first); - write_cbor(el.second); + write_cbor(el.second, depth + 1); } break; } @@ -383,11 +399,22 @@ class binary_writer } /*! - @param[in] j JSON value to serialize + @param[in] j JSON value to serialize + @param[in] depth nesting level of @a j, counted from the top-level value + passed to @ref basic_json::to_msgpack @throw type_error.321 if @a j or a value nested in it is discarded + + @sa @ref write_cbor + @sa https://github.com/nlohmann/json/issues/5392 */ - void write_msgpack(const BasicJsonType& j) + void write_msgpack(const BasicJsonType& j, const std::size_t depth = 0) { + if (JSON_HEDLEY_UNLIKELY(depth >= recursion_depth_limit()) && (j.is_array() || j.is_object())) + { + write_msgpack_iterative(j); + return; + } + switch (j.type()) { case value_t::null: // nil @@ -503,29 +530,11 @@ class binary_writer case value_t::array: { // step 1: write control byte and the array size - const auto N = to_msgpack_length(j.m_data.m_value.array->size(), j); - if (N <= 15) - { - // fixarray - write_number(static_cast(0x90 | N)); - } - else if (N <= (std::numeric_limits::max)()) - { - // array 16 - oa.write_character(to_char_type(0xDC)); - write_number(static_cast(N)); - } - else - { - // array 32 - oa.write_character(to_char_type(0xDD)); - write_number(static_cast(N)); - } + write_msgpack_array_prefix(j.m_data.m_value.array->size(), j); - // step 2: write each element for (const auto& el : *j.m_data.m_value.array) { - write_msgpack(el); + write_msgpack(el, depth + 1); } break; } @@ -621,26 +630,8 @@ class binary_writer case value_t::object: { // step 1: write control byte and the object size - const auto N = to_msgpack_length(j.m_data.m_value.object->size(), j); - if (N <= 15) - { - // fixmap - write_number(static_cast(0x80 | (N & 0xF))); - } - else if (N <= (std::numeric_limits::max)()) - { - // map 16 - oa.write_character(to_char_type(0xDE)); - write_number(static_cast(N)); - } - else - { - // map 32 - oa.write_character(to_char_type(0xDF)); - write_number(static_cast(N)); - } + write_msgpack_object_prefix(j.m_data.m_value.object->size(), j); - // step 2: write each element for (const auto& el : *j.m_data.m_value.object) { // as in write_cbor, el.first is checked here against the @@ -651,7 +642,7 @@ class binary_writer check_utf8(el.first, j); } write_msgpack(el.first); - write_msgpack(el.second); + write_msgpack(el.second, depth + 1); } break; } @@ -669,14 +660,26 @@ class binary_writer @param[in] add_prefix whether prefixes need to be used for this value @param[in] use_bjdata whether write in BJData format, default is false @param[in] bjdata_version which BJData version to use, default is draft2 + @param[in] depth nesting level of @a j, counted from the top-level value + passed to @ref basic_json::to_ubjson or @ref basic_json::to_bjdata @throw type_error.316 if a string value or an object key is not valid UTF-8 @throw type_error.321 if @a j or a value nested in it is discarded + + @sa @ref write_cbor + @sa https://github.com/nlohmann/json/issues/5392 */ void write_ubjson(const BasicJsonType& j, const bool use_count, const bool use_type, const bool add_prefix = true, - const bool use_bjdata = false, const bjdata_version_t bjdata_version = bjdata_version_t::draft2) + const bool use_bjdata = false, const bjdata_version_t bjdata_version = bjdata_version_t::draft2, + const std::size_t depth = 0) { + if (JSON_HEDLEY_UNLIKELY(depth >= recursion_depth_limit()) && (j.is_array() || j.is_object())) + { + write_ubjson_iterative(j, use_count, use_type, add_prefix, use_bjdata, bjdata_version); + return; + } + const bool bjdata_draft3 = use_bjdata && bjdata_version == bjdata_version_t::draft3; switch (j.type()) @@ -737,55 +740,15 @@ class binary_writer case value_t::array: { - if (add_prefix) - { - oa.write_character(to_char_type('[')); - } - bool prefix_required = true; - if (use_type && !j.m_data.m_value.array->empty()) - { - if (!use_count) - { - JSON_THROW(other_error::create(502, "use_type requires use_size = true", &j)); - } - const CharType first_prefix = ubjson_prefix(j.front(), use_bjdata); - const bool same_prefix = std::all_of(j.begin() + 1, j.end(), - [this, first_prefix, use_bjdata](const BasicJsonType & v) - { - return ubjson_prefix(v, use_bjdata) == first_prefix; - }); - - // an optimized array of a valueless type carries no payload, so a - // reader has nothing but the declared count to bound the allocation - // by and refuses an excessive one. Write the unoptimized form for - // those, at one byte per element, so the result can be read back. - // Objects are not affected: every element is preceded by its key. - const bool valueless_type = (first_prefix == 'Z' || first_prefix == 'T' || first_prefix == 'F'); - const bool excessive_valueless = valueless_type - && j.m_data.m_value.array->size() > detail::max_valueless_container_size; - - if (same_prefix && !excessive_valueless - && !(use_bjdata && is_bjdata_excluded_type_marker(first_prefix))) - { - prefix_required = false; - oa.write_character(to_char_type('$')); - oa.write_character(first_prefix); - } - } - - if (use_count) - { - oa.write_character(to_char_type('#')); - write_number_with_ubjson_prefix(j.m_data.m_value.array->size(), true, use_bjdata); - } + const bool write_closer = write_ubjson_start_array(j, use_count, use_type, add_prefix, use_bjdata, prefix_required); for (const auto& el : *j.m_data.m_value.array) { - write_ubjson(el, use_count, use_type, prefix_required, use_bjdata, bjdata_version); + write_ubjson(el, use_count, use_type, prefix_required, use_bjdata, bjdata_version, depth + 1); } - if (!use_count) + if (write_closer) { oa.write_character(to_char_type(']')); } @@ -843,7 +806,7 @@ class binary_writer case value_t::object: { - if (use_bjdata && j.m_data.m_value.object->size() == 3 && j.m_data.m_value.object->find("_ArrayType_") != j.m_data.m_value.object->end() && j.m_data.m_value.object->find("_ArraySize_") != j.m_data.m_value.object->end() && j.m_data.m_value.object->find("_ArrayData_") != j.m_data.m_value.object->end()) + if (use_bjdata && is_bjdata_ndarray(j)) { if (!write_bjdata_ndarray(*j.m_data.m_value.object, use_count, use_type, bjdata_version)) // decode bjdata ndarray in the JData format (https://github.com/NeuroJSON/jdata) { @@ -851,38 +814,8 @@ class binary_writer } } - if (add_prefix) - { - oa.write_character(to_char_type('{')); - } - bool prefix_required = true; - if (use_type && !j.m_data.m_value.object->empty()) - { - if (!use_count) - { - JSON_THROW(other_error::create(502, "use_type requires use_size = true", &j)); - } - const CharType first_prefix = ubjson_prefix(j.front(), use_bjdata); - const bool same_prefix = std::all_of(j.begin(), j.end(), - [this, first_prefix, use_bjdata](const BasicJsonType & v) - { - return ubjson_prefix(v, use_bjdata) == first_prefix; - }); - - if (same_prefix && !(use_bjdata && is_bjdata_excluded_type_marker(first_prefix))) - { - prefix_required = false; - oa.write_character(to_char_type('$')); - oa.write_character(first_prefix); - } - } - - if (use_count) - { - oa.write_character(to_char_type('#')); - write_number_with_ubjson_prefix(j.m_data.m_value.object->size(), true, use_bjdata); - } + const bool write_closer = write_ubjson_start_object(j, use_count, use_type, add_prefix, use_bjdata, prefix_required); for (const auto& el : *j.m_data.m_value.object) { @@ -892,10 +825,10 @@ class binary_writer oa.write_characters( reinterpret_cast(key.data()), key.size()); - write_ubjson(el.second, use_count, use_type, prefix_required, use_bjdata, bjdata_version); + write_ubjson(el.second, use_count, use_type, prefix_required, use_bjdata, bjdata_version, depth + 1); } - if (!use_count) + if (write_closer) { oa.write_character(to_char_type('}')); } @@ -936,6 +869,517 @@ class binary_writer JSON_THROW(type_error::create(321, concat("cannot serialize discarded value to ", format_name), &j)); } + void write_msgpack_array_prefix(const std::size_t N, const BasicJsonType& j) + { + const auto n = to_msgpack_length(N, j); + if (n <= 15) + { + // fixarray + write_number(static_cast(0x90 | n)); + } + else if (n <= (std::numeric_limits::max)()) + { + // array 16 + oa.write_character(to_char_type(0xDC)); + write_number(static_cast(n)); + } + else + { + // array 32 + oa.write_character(to_char_type(0xDD)); + write_number(static_cast(n)); + } + } + + void write_msgpack_object_prefix(const std::size_t N, const BasicJsonType& j) + { + const auto n = to_msgpack_length(N, j); + if (n <= 15) + { + // fixmap + write_number(static_cast(0x80 | (n & 0xF))); + } + else if (n <= (std::numeric_limits::max)()) + { + // map 16 + oa.write_character(to_char_type(0xDE)); + write_number(static_cast(n)); + } + else + { + // map 32 + oa.write_character(to_char_type(0xDF)); + write_number(static_cast(n)); + } + } + + /// @brief a CBOR or MessagePack array or object whose elements + /// @ref write_cbor_iterative or @ref write_msgpack_iterative is + /// still writing + struct binary_container_frame + { + explicit binary_container_frame(const BasicJsonType* value_) noexcept + : value(value_) + { + if (value->is_object()) + { + object_it = value->m_data.m_value.object->cbegin(); + } + else + { + array_it = value->m_data.m_value.array->cbegin(); + } + } + + // declared for GCC's -Weffc++, which asks for them in a class with + // pointer members and a non-trivial destructor; the exception + // specifications are left implicit, as GCC 4.8 rejects explicit ones + // that differ from them + binary_container_frame(const binary_container_frame&) = default; + binary_container_frame(binary_container_frame&&) = default; + binary_container_frame& operator=(const binary_container_frame&) = default; + binary_container_frame& operator=(binary_container_frame&&) = default; + ~binary_container_frame() = default; + + /// the array or object being written + const BasicJsonType* value; + /// value's elements still to write; which of the two is live follows + /// from the type of value. They are kept side by side rather than in + /// a union, which would need its special members written out by + /// hand, see detail/iterators/internal_iterator.hpp + typename BasicJsonType::object_t::const_iterator object_it{}; + typename BasicJsonType::array_t::const_iterator array_it{}; + }; + + /*! + @brief write @a j with @ref write_cbor, or write its header and push a + frame for @ref write_cbor_iterative to continue with its elements + + A scalar, and an empty array or object, are written out in full: there is + nothing below them for @ref write_cbor_iterative to come back to, so + nothing is pushed for them. + */ + void write_cbor_value_or_push(const BasicJsonType& j, std::vector& stack) + { + if (j.is_array()) + { + write_cbor_head(0x80, j.m_data.m_value.array->size()); + if (!j.m_data.m_value.array->empty()) + { + stack.emplace_back(&j); + } + return; + } + + if (j.is_object()) + { + write_cbor_head(0xA0, j.m_data.m_value.object->size()); + if (!j.m_data.m_value.object->empty()) + { + stack.emplace_back(&j); + } + return; + } + + write_cbor(j); + } + + /*! + @brief write out @a root and everything below it without the call stack + + Emits the same bytes as @ref write_cbor, keeping the containers it has + entered on an explicit stack instead of descending into them. Only reached + for values nested deeper than @ref recursion_depth_limit, which is why it + is not written for speed. + */ + void write_cbor_iterative(const BasicJsonType& root) + { + // only a container with elements is ever pushed; see write_cbor_value_or_push + std::vector stack; + write_cbor_value_or_push(root, stack); + + while (!stack.empty()) + { + const binary_container_frame current = stack.back(); + + if (current.value->is_array()) + { + const auto& array = *current.value->m_data.m_value.array; + if (current.array_it == array.cend()) + { + stack.pop_back(); + continue; + } + + // read the child before pushing: entering it can move every frame + const BasicJsonType* child = &(*current.array_it); + ++stack.back().array_it; + write_cbor_value_or_push(*child, stack); + } + else + { + const auto& object = *current.value->m_data.m_value.object; + if (current.object_it == object.cend()) + { + stack.pop_back(); + continue; + } + + // el.first is checked here, against the object as diagnostics + // context, like the matching check in write_cbor's object case + if (error_handler == error_handler_t::strict) + { + check_utf8(current.object_it->first, *current.value); + } + write_cbor(current.object_it->first); + const BasicJsonType* child = &(current.object_it->second); + ++stack.back().object_it; + write_cbor_value_or_push(*child, stack); + } + } + } + + /*! + @brief write @a j with @ref write_msgpack, or write its header and push a + frame for @ref write_msgpack_iterative to continue with its elements + + @sa @ref write_cbor_value_or_push + */ + void write_msgpack_value_or_push(const BasicJsonType& j, std::vector& stack) + { + if (j.is_array()) + { + write_msgpack_array_prefix(j.m_data.m_value.array->size(), j); + if (!j.m_data.m_value.array->empty()) + { + stack.emplace_back(&j); + } + return; + } + + if (j.is_object()) + { + write_msgpack_object_prefix(j.m_data.m_value.object->size(), j); + if (!j.m_data.m_value.object->empty()) + { + stack.emplace_back(&j); + } + return; + } + + write_msgpack(j); + } + + /*! + @brief write out @a root and everything below it without the call stack + + @sa @ref write_cbor_iterative + */ + void write_msgpack_iterative(const BasicJsonType& root) + { + std::vector stack; + write_msgpack_value_or_push(root, stack); + + while (!stack.empty()) + { + const binary_container_frame current = stack.back(); + + if (current.value->is_array()) + { + const auto& array = *current.value->m_data.m_value.array; + if (current.array_it == array.cend()) + { + stack.pop_back(); + continue; + } + + const BasicJsonType* child = &(*current.array_it); + ++stack.back().array_it; + write_msgpack_value_or_push(*child, stack); + } + else + { + const auto& object = *current.value->m_data.m_value.object; + if (current.object_it == object.cend()) + { + stack.pop_back(); + continue; + } + + if (error_handler == error_handler_t::strict) + { + check_utf8(current.object_it->first, *current.value); + } + write_msgpack(current.object_it->first); + const BasicJsonType* child = &(current.object_it->second); + ++stack.back().object_it; + write_msgpack_value_or_push(*child, stack); + } + } + } + + /// @return true when a closing ']' still has to be written after the elements + bool write_ubjson_start_array(const BasicJsonType& j, const bool use_count, const bool use_type, + const bool add_prefix, const bool use_bjdata, bool& prefix_required) + { + prefix_required = true; + if (add_prefix) + { + oa.write_character(to_char_type('[')); + } + + if (use_type && !j.m_data.m_value.array->empty()) + { + if (!use_count) + { + JSON_THROW(other_error::create(502, "use_type requires use_size = true", &j)); + } + const CharType first_prefix = ubjson_prefix(j.front(), use_bjdata); + const bool same_prefix = std::all_of(j.begin() + 1, j.end(), + [this, first_prefix, use_bjdata](const BasicJsonType & v) + { + return ubjson_prefix(v, use_bjdata) == first_prefix; + }); + + // an optimized array of a valueless type carries no payload, so a + // reader has nothing but the declared count to bound the allocation + // by and refuses an excessive one. Write the unoptimized form for + // those, at one byte per element, so the result can be read back. + // Objects are not affected: every element is preceded by its key. + const bool valueless_type = (first_prefix == 'Z' || first_prefix == 'T' || first_prefix == 'F'); + const bool excessive_valueless = valueless_type + && j.m_data.m_value.array->size() > detail::max_valueless_container_size; + + if (same_prefix && !excessive_valueless + && !(use_bjdata && is_bjdata_excluded_type_marker(first_prefix))) + { + prefix_required = false; + oa.write_character(to_char_type('$')); + oa.write_character(first_prefix); + } + } + + if (use_count) + { + oa.write_character(to_char_type('#')); + write_number_with_ubjson_prefix(j.m_data.m_value.array->size(), true, use_bjdata); + } + + return !use_count; + } + + /// @return true when a closing '}' still has to be written after the elements + bool write_ubjson_start_object(const BasicJsonType& j, const bool use_count, const bool use_type, + const bool add_prefix, const bool use_bjdata, bool& prefix_required) + { + prefix_required = true; + if (add_prefix) + { + oa.write_character(to_char_type('{')); + } + + if (use_type && !j.m_data.m_value.object->empty()) + { + if (!use_count) + { + JSON_THROW(other_error::create(502, "use_type requires use_size = true", &j)); + } + const CharType first_prefix = ubjson_prefix(j.front(), use_bjdata); + const bool same_prefix = std::all_of(j.begin(), j.end(), + [this, first_prefix, use_bjdata](const BasicJsonType & v) + { + return ubjson_prefix(v, use_bjdata) == first_prefix; + }); + + if (same_prefix && !(use_bjdata && is_bjdata_excluded_type_marker(first_prefix))) + { + prefix_required = false; + oa.write_character(to_char_type('$')); + oa.write_character(first_prefix); + } + } + + if (use_count) + { + oa.write_character(to_char_type('#')); + write_number_with_ubjson_prefix(j.m_data.m_value.object->size(), true, use_bjdata); + } + + return !use_count; + } + + /*! + @brief whether @a j is a BJData ND-array annotation object + (https://github.com/NeuroJSON/jdata) + + Used by both the recursive object case of @ref write_ubjson and + @ref write_ubjson_value_or_push, which must agree on what counts as an + ND-array: @a j is only actually written as one once @ref + write_bjdata_ndarray has also accepted its contents. + + @pre @a j.is_object() + */ + static bool is_bjdata_ndarray(const BasicJsonType& j) + { + const auto& object = *j.m_data.m_value.object; + return object.size() == 3 + && object.find("_ArrayType_") != object.end() + && object.find("_ArraySize_") != object.end() + && object.find("_ArrayData_") != object.end(); + } + + /// @brief an object or array @ref write_ubjson_iterative is still writing + /// the elements of + struct ubjson_frame + { + ubjson_frame(const BasicJsonType* value_, const bool prefix_required_) noexcept + : value(value_) + , prefix_required(prefix_required_) + { + if (value->is_object()) + { + object_it = value->m_data.m_value.object->cbegin(); + } + else + { + array_it = value->m_data.m_value.array->cbegin(); + } + } + + // declared for GCC's -Weffc++, which asks for them in a class with + // pointer members and a non-trivial destructor; the exception + // specifications are left implicit, as GCC 4.8 rejects explicit ones + // that differ from them + ubjson_frame(const ubjson_frame&) = default; + ubjson_frame(ubjson_frame&&) = default; + ubjson_frame& operator=(const ubjson_frame&) = default; + ubjson_frame& operator=(ubjson_frame&&) = default; + ~ubjson_frame() = default; + + /// the array or object being written + const BasicJsonType* value; + /// whether value's elements each carry their own type marker; an + /// optimized ($type) container writes it once for all of them instead + bool prefix_required; + typename BasicJsonType::object_t::const_iterator object_it{}; + typename BasicJsonType::array_t::const_iterator array_it{}; + }; + + /*! + @brief write @a j with @ref write_ubjson, or write its header and push a + frame for @ref write_ubjson_iterative to continue with its elements + + @param[in] add_prefix whether @a j's own type marker is written now (the + elements of an optimized container, and everything below the + top level, never repeat it) + + @sa @ref write_cbor_value_or_push + */ + void write_ubjson_value_or_push(const BasicJsonType& j, const bool add_prefix, const bool use_count, + const bool use_type, const bool use_bjdata, const bjdata_version_t bjdata_version, + std::vector& stack) + { + if (!j.is_array() && !j.is_object()) + { + write_ubjson(j, use_count, use_type, add_prefix, use_bjdata, bjdata_version); + return; + } + + if (use_bjdata && j.is_object() && is_bjdata_ndarray(j) + && !write_bjdata_ndarray(*j.m_data.m_value.object, use_count, use_type, bjdata_version)) + { + // fully written as an ND-array: nothing below it to come back to + return; + } + + const bool is_array = j.is_array(); + bool prefix_required = true; + if (is_array) + { + write_ubjson_start_array(j, use_count, use_type, add_prefix, use_bjdata, prefix_required); + } + else + { + write_ubjson_start_object(j, use_count, use_type, add_prefix, use_bjdata, prefix_required); + } + + const bool empty = is_array ? j.m_data.m_value.array->empty() : j.m_data.m_value.object->empty(); + if (!empty) + { + stack.emplace_back(&j, prefix_required); + return; + } + + // write_ubjson_start_array/_object return !use_count, i.e. whether a + // closer still has to be written; use_count is constant for the whole + // document, so that is recomputed here instead of being carried along + if (!use_count) + { + oa.write_character(to_char_type(is_array ? ']' : '}')); + } + } + + /*! + @brief write out @a root and everything below it without the call stack + + @sa @ref write_cbor_iterative + */ + void write_ubjson_iterative(const BasicJsonType& root, const bool use_count, const bool use_type, + const bool add_prefix, const bool use_bjdata, const bjdata_version_t bjdata_version) + { + std::vector stack; + write_ubjson_value_or_push(root, add_prefix, use_count, use_type, use_bjdata, bjdata_version, stack); + + while (!stack.empty()) + { + const ubjson_frame current = stack.back(); + const BasicJsonType& j = *current.value; + + if (j.is_array()) + { + const auto& array = *j.m_data.m_value.array; + if (current.array_it == array.cend()) + { + if (!use_count) + { + oa.write_character(to_char_type(']')); + } + stack.pop_back(); + continue; + } + + const BasicJsonType* child = &(*current.array_it); + const bool child_prefix = current.prefix_required; + ++stack.back().array_it; + write_ubjson_value_or_push(*child, child_prefix, use_count, use_type, use_bjdata, bjdata_version, stack); + } + else + { + const auto& object = *j.m_data.m_value.object; + if (current.object_it == object.cend()) + { + if (!use_count) + { + oa.write_character(to_char_type('}')); + } + stack.pop_back(); + continue; + } + + string_t storage; + const string_t& key = sanitize_utf8_for_write(current.object_it->first, j, storage); + write_number_with_ubjson_prefix(key.size(), true, use_bjdata); + oa.write_characters( + reinterpret_cast(key.data()), + key.size()); + const BasicJsonType* child = &(current.object_it->second); + const bool child_prefix = current.prefix_required; + ++stack.back().object_it; + write_ubjson_value_or_push(*child, child_prefix, use_count, use_type, use_bjdata, bjdata_version, stack); + } + } + } + ////////// // BSON // ////////// diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 06908b02b..0119b01cc 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -21518,6 +21518,8 @@ class output_adapter } // namespace detail NLOHMANN_JSON_NAMESPACE_END +// #include + // #include // #include @@ -21648,13 +21650,30 @@ class binary_writer } /*! - @param[in] j JSON value to serialize + @param[in] j JSON value to serialize + @param[in] depth nesting level of @a j, counted from the top-level value + passed to @ref basic_json::to_cbor @throw type_error.316 if a string value or an object key is not valid UTF-8 @throw type_error.321 if @a j or a value nested in it is discarded + + Serializing a container descends into its elements, so a value nested deeply + enough used to exhaust the call stack and terminate the process with no + exception to catch. The descent is bounded here: once @ref recursion_depth_limit + levels have been entered, @ref write_cbor_iterative writes out what is left + without the call stack. A value nested less deeply than that - all but a + vanishing minority - is written by exactly the code that always wrote it. + + @sa https://github.com/nlohmann/json/issues/5392 */ - void write_cbor(const BasicJsonType& j) + void write_cbor(const BasicJsonType& j, const std::size_t depth = 0) { + if (JSON_HEDLEY_UNLIKELY(depth >= recursion_depth_limit()) && (j.is_array() || j.is_object())) + { + write_cbor_iterative(j); + return; + } + switch (j.type()) { case value_t::null: @@ -21736,10 +21755,9 @@ class binary_writer // step 1: write control byte and the array size write_cbor_head(0x80, j.m_data.m_value.array->size()); - // step 2: write each element for (const auto& el : *j.m_data.m_value.array) { - write_cbor(el); + write_cbor(el, depth + 1); } break; } @@ -21794,7 +21812,6 @@ class binary_writer // step 1: write control byte and the object size write_cbor_head(0xA0, j.m_data.m_value.object->size()); - // step 2: write each element for (const auto& el : *j.m_data.m_value.object) { // el.first is checked here, against the object as @@ -21809,7 +21826,7 @@ class binary_writer check_utf8(el.first, j); } write_cbor(el.first); - write_cbor(el.second); + write_cbor(el.second, depth + 1); } break; } @@ -21875,11 +21892,22 @@ class binary_writer } /*! - @param[in] j JSON value to serialize + @param[in] j JSON value to serialize + @param[in] depth nesting level of @a j, counted from the top-level value + passed to @ref basic_json::to_msgpack @throw type_error.321 if @a j or a value nested in it is discarded + + @sa @ref write_cbor + @sa https://github.com/nlohmann/json/issues/5392 */ - void write_msgpack(const BasicJsonType& j) + void write_msgpack(const BasicJsonType& j, const std::size_t depth = 0) { + if (JSON_HEDLEY_UNLIKELY(depth >= recursion_depth_limit()) && (j.is_array() || j.is_object())) + { + write_msgpack_iterative(j); + return; + } + switch (j.type()) { case value_t::null: // nil @@ -21995,29 +22023,11 @@ class binary_writer case value_t::array: { // step 1: write control byte and the array size - const auto N = to_msgpack_length(j.m_data.m_value.array->size(), j); - if (N <= 15) - { - // fixarray - write_number(static_cast(0x90 | N)); - } - else if (N <= (std::numeric_limits::max)()) - { - // array 16 - oa.write_character(to_char_type(0xDC)); - write_number(static_cast(N)); - } - else - { - // array 32 - oa.write_character(to_char_type(0xDD)); - write_number(static_cast(N)); - } + write_msgpack_array_prefix(j.m_data.m_value.array->size(), j); - // step 2: write each element for (const auto& el : *j.m_data.m_value.array) { - write_msgpack(el); + write_msgpack(el, depth + 1); } break; } @@ -22113,26 +22123,8 @@ class binary_writer case value_t::object: { // step 1: write control byte and the object size - const auto N = to_msgpack_length(j.m_data.m_value.object->size(), j); - if (N <= 15) - { - // fixmap - write_number(static_cast(0x80 | (N & 0xF))); - } - else if (N <= (std::numeric_limits::max)()) - { - // map 16 - oa.write_character(to_char_type(0xDE)); - write_number(static_cast(N)); - } - else - { - // map 32 - oa.write_character(to_char_type(0xDF)); - write_number(static_cast(N)); - } + write_msgpack_object_prefix(j.m_data.m_value.object->size(), j); - // step 2: write each element for (const auto& el : *j.m_data.m_value.object) { // as in write_cbor, el.first is checked here against the @@ -22143,7 +22135,7 @@ class binary_writer check_utf8(el.first, j); } write_msgpack(el.first); - write_msgpack(el.second); + write_msgpack(el.second, depth + 1); } break; } @@ -22161,14 +22153,26 @@ class binary_writer @param[in] add_prefix whether prefixes need to be used for this value @param[in] use_bjdata whether write in BJData format, default is false @param[in] bjdata_version which BJData version to use, default is draft2 + @param[in] depth nesting level of @a j, counted from the top-level value + passed to @ref basic_json::to_ubjson or @ref basic_json::to_bjdata @throw type_error.316 if a string value or an object key is not valid UTF-8 @throw type_error.321 if @a j or a value nested in it is discarded + + @sa @ref write_cbor + @sa https://github.com/nlohmann/json/issues/5392 */ void write_ubjson(const BasicJsonType& j, const bool use_count, const bool use_type, const bool add_prefix = true, - const bool use_bjdata = false, const bjdata_version_t bjdata_version = bjdata_version_t::draft2) + const bool use_bjdata = false, const bjdata_version_t bjdata_version = bjdata_version_t::draft2, + const std::size_t depth = 0) { + if (JSON_HEDLEY_UNLIKELY(depth >= recursion_depth_limit()) && (j.is_array() || j.is_object())) + { + write_ubjson_iterative(j, use_count, use_type, add_prefix, use_bjdata, bjdata_version); + return; + } + const bool bjdata_draft3 = use_bjdata && bjdata_version == bjdata_version_t::draft3; switch (j.type()) @@ -22229,55 +22233,15 @@ class binary_writer case value_t::array: { - if (add_prefix) - { - oa.write_character(to_char_type('[')); - } - bool prefix_required = true; - if (use_type && !j.m_data.m_value.array->empty()) - { - if (!use_count) - { - JSON_THROW(other_error::create(502, "use_type requires use_size = true", &j)); - } - const CharType first_prefix = ubjson_prefix(j.front(), use_bjdata); - const bool same_prefix = std::all_of(j.begin() + 1, j.end(), - [this, first_prefix, use_bjdata](const BasicJsonType & v) - { - return ubjson_prefix(v, use_bjdata) == first_prefix; - }); - - // an optimized array of a valueless type carries no payload, so a - // reader has nothing but the declared count to bound the allocation - // by and refuses an excessive one. Write the unoptimized form for - // those, at one byte per element, so the result can be read back. - // Objects are not affected: every element is preceded by its key. - const bool valueless_type = (first_prefix == 'Z' || first_prefix == 'T' || first_prefix == 'F'); - const bool excessive_valueless = valueless_type - && j.m_data.m_value.array->size() > detail::max_valueless_container_size; - - if (same_prefix && !excessive_valueless - && !(use_bjdata && is_bjdata_excluded_type_marker(first_prefix))) - { - prefix_required = false; - oa.write_character(to_char_type('$')); - oa.write_character(first_prefix); - } - } - - if (use_count) - { - oa.write_character(to_char_type('#')); - write_number_with_ubjson_prefix(j.m_data.m_value.array->size(), true, use_bjdata); - } + const bool write_closer = write_ubjson_start_array(j, use_count, use_type, add_prefix, use_bjdata, prefix_required); for (const auto& el : *j.m_data.m_value.array) { - write_ubjson(el, use_count, use_type, prefix_required, use_bjdata, bjdata_version); + write_ubjson(el, use_count, use_type, prefix_required, use_bjdata, bjdata_version, depth + 1); } - if (!use_count) + if (write_closer) { oa.write_character(to_char_type(']')); } @@ -22335,7 +22299,7 @@ class binary_writer case value_t::object: { - if (use_bjdata && j.m_data.m_value.object->size() == 3 && j.m_data.m_value.object->find("_ArrayType_") != j.m_data.m_value.object->end() && j.m_data.m_value.object->find("_ArraySize_") != j.m_data.m_value.object->end() && j.m_data.m_value.object->find("_ArrayData_") != j.m_data.m_value.object->end()) + if (use_bjdata && is_bjdata_ndarray(j)) { if (!write_bjdata_ndarray(*j.m_data.m_value.object, use_count, use_type, bjdata_version)) // decode bjdata ndarray in the JData format (https://github.com/NeuroJSON/jdata) { @@ -22343,38 +22307,8 @@ class binary_writer } } - if (add_prefix) - { - oa.write_character(to_char_type('{')); - } - bool prefix_required = true; - if (use_type && !j.m_data.m_value.object->empty()) - { - if (!use_count) - { - JSON_THROW(other_error::create(502, "use_type requires use_size = true", &j)); - } - const CharType first_prefix = ubjson_prefix(j.front(), use_bjdata); - const bool same_prefix = std::all_of(j.begin(), j.end(), - [this, first_prefix, use_bjdata](const BasicJsonType & v) - { - return ubjson_prefix(v, use_bjdata) == first_prefix; - }); - - if (same_prefix && !(use_bjdata && is_bjdata_excluded_type_marker(first_prefix))) - { - prefix_required = false; - oa.write_character(to_char_type('$')); - oa.write_character(first_prefix); - } - } - - if (use_count) - { - oa.write_character(to_char_type('#')); - write_number_with_ubjson_prefix(j.m_data.m_value.object->size(), true, use_bjdata); - } + const bool write_closer = write_ubjson_start_object(j, use_count, use_type, add_prefix, use_bjdata, prefix_required); for (const auto& el : *j.m_data.m_value.object) { @@ -22384,10 +22318,10 @@ class binary_writer oa.write_characters( reinterpret_cast(key.data()), key.size()); - write_ubjson(el.second, use_count, use_type, prefix_required, use_bjdata, bjdata_version); + write_ubjson(el.second, use_count, use_type, prefix_required, use_bjdata, bjdata_version, depth + 1); } - if (!use_count) + if (write_closer) { oa.write_character(to_char_type('}')); } @@ -22428,6 +22362,517 @@ class binary_writer JSON_THROW(type_error::create(321, concat("cannot serialize discarded value to ", format_name), &j)); } + void write_msgpack_array_prefix(const std::size_t N, const BasicJsonType& j) + { + const auto n = to_msgpack_length(N, j); + if (n <= 15) + { + // fixarray + write_number(static_cast(0x90 | n)); + } + else if (n <= (std::numeric_limits::max)()) + { + // array 16 + oa.write_character(to_char_type(0xDC)); + write_number(static_cast(n)); + } + else + { + // array 32 + oa.write_character(to_char_type(0xDD)); + write_number(static_cast(n)); + } + } + + void write_msgpack_object_prefix(const std::size_t N, const BasicJsonType& j) + { + const auto n = to_msgpack_length(N, j); + if (n <= 15) + { + // fixmap + write_number(static_cast(0x80 | (n & 0xF))); + } + else if (n <= (std::numeric_limits::max)()) + { + // map 16 + oa.write_character(to_char_type(0xDE)); + write_number(static_cast(n)); + } + else + { + // map 32 + oa.write_character(to_char_type(0xDF)); + write_number(static_cast(n)); + } + } + + /// @brief a CBOR or MessagePack array or object whose elements + /// @ref write_cbor_iterative or @ref write_msgpack_iterative is + /// still writing + struct binary_container_frame + { + explicit binary_container_frame(const BasicJsonType* value_) noexcept + : value(value_) + { + if (value->is_object()) + { + object_it = value->m_data.m_value.object->cbegin(); + } + else + { + array_it = value->m_data.m_value.array->cbegin(); + } + } + + // declared for GCC's -Weffc++, which asks for them in a class with + // pointer members and a non-trivial destructor; the exception + // specifications are left implicit, as GCC 4.8 rejects explicit ones + // that differ from them + binary_container_frame(const binary_container_frame&) = default; + binary_container_frame(binary_container_frame&&) = default; + binary_container_frame& operator=(const binary_container_frame&) = default; + binary_container_frame& operator=(binary_container_frame&&) = default; + ~binary_container_frame() = default; + + /// the array or object being written + const BasicJsonType* value; + /// value's elements still to write; which of the two is live follows + /// from the type of value. They are kept side by side rather than in + /// a union, which would need its special members written out by + /// hand, see detail/iterators/internal_iterator.hpp + typename BasicJsonType::object_t::const_iterator object_it{}; + typename BasicJsonType::array_t::const_iterator array_it{}; + }; + + /*! + @brief write @a j with @ref write_cbor, or write its header and push a + frame for @ref write_cbor_iterative to continue with its elements + + A scalar, and an empty array or object, are written out in full: there is + nothing below them for @ref write_cbor_iterative to come back to, so + nothing is pushed for them. + */ + void write_cbor_value_or_push(const BasicJsonType& j, std::vector& stack) + { + if (j.is_array()) + { + write_cbor_head(0x80, j.m_data.m_value.array->size()); + if (!j.m_data.m_value.array->empty()) + { + stack.emplace_back(&j); + } + return; + } + + if (j.is_object()) + { + write_cbor_head(0xA0, j.m_data.m_value.object->size()); + if (!j.m_data.m_value.object->empty()) + { + stack.emplace_back(&j); + } + return; + } + + write_cbor(j); + } + + /*! + @brief write out @a root and everything below it without the call stack + + Emits the same bytes as @ref write_cbor, keeping the containers it has + entered on an explicit stack instead of descending into them. Only reached + for values nested deeper than @ref recursion_depth_limit, which is why it + is not written for speed. + */ + void write_cbor_iterative(const BasicJsonType& root) + { + // only a container with elements is ever pushed; see write_cbor_value_or_push + std::vector stack; + write_cbor_value_or_push(root, stack); + + while (!stack.empty()) + { + const binary_container_frame current = stack.back(); + + if (current.value->is_array()) + { + const auto& array = *current.value->m_data.m_value.array; + if (current.array_it == array.cend()) + { + stack.pop_back(); + continue; + } + + // read the child before pushing: entering it can move every frame + const BasicJsonType* child = &(*current.array_it); + ++stack.back().array_it; + write_cbor_value_or_push(*child, stack); + } + else + { + const auto& object = *current.value->m_data.m_value.object; + if (current.object_it == object.cend()) + { + stack.pop_back(); + continue; + } + + // el.first is checked here, against the object as diagnostics + // context, like the matching check in write_cbor's object case + if (error_handler == error_handler_t::strict) + { + check_utf8(current.object_it->first, *current.value); + } + write_cbor(current.object_it->first); + const BasicJsonType* child = &(current.object_it->second); + ++stack.back().object_it; + write_cbor_value_or_push(*child, stack); + } + } + } + + /*! + @brief write @a j with @ref write_msgpack, or write its header and push a + frame for @ref write_msgpack_iterative to continue with its elements + + @sa @ref write_cbor_value_or_push + */ + void write_msgpack_value_or_push(const BasicJsonType& j, std::vector& stack) + { + if (j.is_array()) + { + write_msgpack_array_prefix(j.m_data.m_value.array->size(), j); + if (!j.m_data.m_value.array->empty()) + { + stack.emplace_back(&j); + } + return; + } + + if (j.is_object()) + { + write_msgpack_object_prefix(j.m_data.m_value.object->size(), j); + if (!j.m_data.m_value.object->empty()) + { + stack.emplace_back(&j); + } + return; + } + + write_msgpack(j); + } + + /*! + @brief write out @a root and everything below it without the call stack + + @sa @ref write_cbor_iterative + */ + void write_msgpack_iterative(const BasicJsonType& root) + { + std::vector stack; + write_msgpack_value_or_push(root, stack); + + while (!stack.empty()) + { + const binary_container_frame current = stack.back(); + + if (current.value->is_array()) + { + const auto& array = *current.value->m_data.m_value.array; + if (current.array_it == array.cend()) + { + stack.pop_back(); + continue; + } + + const BasicJsonType* child = &(*current.array_it); + ++stack.back().array_it; + write_msgpack_value_or_push(*child, stack); + } + else + { + const auto& object = *current.value->m_data.m_value.object; + if (current.object_it == object.cend()) + { + stack.pop_back(); + continue; + } + + if (error_handler == error_handler_t::strict) + { + check_utf8(current.object_it->first, *current.value); + } + write_msgpack(current.object_it->first); + const BasicJsonType* child = &(current.object_it->second); + ++stack.back().object_it; + write_msgpack_value_or_push(*child, stack); + } + } + } + + /// @return true when a closing ']' still has to be written after the elements + bool write_ubjson_start_array(const BasicJsonType& j, const bool use_count, const bool use_type, + const bool add_prefix, const bool use_bjdata, bool& prefix_required) + { + prefix_required = true; + if (add_prefix) + { + oa.write_character(to_char_type('[')); + } + + if (use_type && !j.m_data.m_value.array->empty()) + { + if (!use_count) + { + JSON_THROW(other_error::create(502, "use_type requires use_size = true", &j)); + } + const CharType first_prefix = ubjson_prefix(j.front(), use_bjdata); + const bool same_prefix = std::all_of(j.begin() + 1, j.end(), + [this, first_prefix, use_bjdata](const BasicJsonType & v) + { + return ubjson_prefix(v, use_bjdata) == first_prefix; + }); + + // an optimized array of a valueless type carries no payload, so a + // reader has nothing but the declared count to bound the allocation + // by and refuses an excessive one. Write the unoptimized form for + // those, at one byte per element, so the result can be read back. + // Objects are not affected: every element is preceded by its key. + const bool valueless_type = (first_prefix == 'Z' || first_prefix == 'T' || first_prefix == 'F'); + const bool excessive_valueless = valueless_type + && j.m_data.m_value.array->size() > detail::max_valueless_container_size; + + if (same_prefix && !excessive_valueless + && !(use_bjdata && is_bjdata_excluded_type_marker(first_prefix))) + { + prefix_required = false; + oa.write_character(to_char_type('$')); + oa.write_character(first_prefix); + } + } + + if (use_count) + { + oa.write_character(to_char_type('#')); + write_number_with_ubjson_prefix(j.m_data.m_value.array->size(), true, use_bjdata); + } + + return !use_count; + } + + /// @return true when a closing '}' still has to be written after the elements + bool write_ubjson_start_object(const BasicJsonType& j, const bool use_count, const bool use_type, + const bool add_prefix, const bool use_bjdata, bool& prefix_required) + { + prefix_required = true; + if (add_prefix) + { + oa.write_character(to_char_type('{')); + } + + if (use_type && !j.m_data.m_value.object->empty()) + { + if (!use_count) + { + JSON_THROW(other_error::create(502, "use_type requires use_size = true", &j)); + } + const CharType first_prefix = ubjson_prefix(j.front(), use_bjdata); + const bool same_prefix = std::all_of(j.begin(), j.end(), + [this, first_prefix, use_bjdata](const BasicJsonType & v) + { + return ubjson_prefix(v, use_bjdata) == first_prefix; + }); + + if (same_prefix && !(use_bjdata && is_bjdata_excluded_type_marker(first_prefix))) + { + prefix_required = false; + oa.write_character(to_char_type('$')); + oa.write_character(first_prefix); + } + } + + if (use_count) + { + oa.write_character(to_char_type('#')); + write_number_with_ubjson_prefix(j.m_data.m_value.object->size(), true, use_bjdata); + } + + return !use_count; + } + + /*! + @brief whether @a j is a BJData ND-array annotation object + (https://github.com/NeuroJSON/jdata) + + Used by both the recursive object case of @ref write_ubjson and + @ref write_ubjson_value_or_push, which must agree on what counts as an + ND-array: @a j is only actually written as one once @ref + write_bjdata_ndarray has also accepted its contents. + + @pre @a j.is_object() + */ + static bool is_bjdata_ndarray(const BasicJsonType& j) + { + const auto& object = *j.m_data.m_value.object; + return object.size() == 3 + && object.find("_ArrayType_") != object.end() + && object.find("_ArraySize_") != object.end() + && object.find("_ArrayData_") != object.end(); + } + + /// @brief an object or array @ref write_ubjson_iterative is still writing + /// the elements of + struct ubjson_frame + { + ubjson_frame(const BasicJsonType* value_, const bool prefix_required_) noexcept + : value(value_) + , prefix_required(prefix_required_) + { + if (value->is_object()) + { + object_it = value->m_data.m_value.object->cbegin(); + } + else + { + array_it = value->m_data.m_value.array->cbegin(); + } + } + + // declared for GCC's -Weffc++, which asks for them in a class with + // pointer members and a non-trivial destructor; the exception + // specifications are left implicit, as GCC 4.8 rejects explicit ones + // that differ from them + ubjson_frame(const ubjson_frame&) = default; + ubjson_frame(ubjson_frame&&) = default; + ubjson_frame& operator=(const ubjson_frame&) = default; + ubjson_frame& operator=(ubjson_frame&&) = default; + ~ubjson_frame() = default; + + /// the array or object being written + const BasicJsonType* value; + /// whether value's elements each carry their own type marker; an + /// optimized ($type) container writes it once for all of them instead + bool prefix_required; + typename BasicJsonType::object_t::const_iterator object_it{}; + typename BasicJsonType::array_t::const_iterator array_it{}; + }; + + /*! + @brief write @a j with @ref write_ubjson, or write its header and push a + frame for @ref write_ubjson_iterative to continue with its elements + + @param[in] add_prefix whether @a j's own type marker is written now (the + elements of an optimized container, and everything below the + top level, never repeat it) + + @sa @ref write_cbor_value_or_push + */ + void write_ubjson_value_or_push(const BasicJsonType& j, const bool add_prefix, const bool use_count, + const bool use_type, const bool use_bjdata, const bjdata_version_t bjdata_version, + std::vector& stack) + { + if (!j.is_array() && !j.is_object()) + { + write_ubjson(j, use_count, use_type, add_prefix, use_bjdata, bjdata_version); + return; + } + + if (use_bjdata && j.is_object() && is_bjdata_ndarray(j) + && !write_bjdata_ndarray(*j.m_data.m_value.object, use_count, use_type, bjdata_version)) + { + // fully written as an ND-array: nothing below it to come back to + return; + } + + const bool is_array = j.is_array(); + bool prefix_required = true; + if (is_array) + { + write_ubjson_start_array(j, use_count, use_type, add_prefix, use_bjdata, prefix_required); + } + else + { + write_ubjson_start_object(j, use_count, use_type, add_prefix, use_bjdata, prefix_required); + } + + const bool empty = is_array ? j.m_data.m_value.array->empty() : j.m_data.m_value.object->empty(); + if (!empty) + { + stack.emplace_back(&j, prefix_required); + return; + } + + // write_ubjson_start_array/_object return !use_count, i.e. whether a + // closer still has to be written; use_count is constant for the whole + // document, so that is recomputed here instead of being carried along + if (!use_count) + { + oa.write_character(to_char_type(is_array ? ']' : '}')); + } + } + + /*! + @brief write out @a root and everything below it without the call stack + + @sa @ref write_cbor_iterative + */ + void write_ubjson_iterative(const BasicJsonType& root, const bool use_count, const bool use_type, + const bool add_prefix, const bool use_bjdata, const bjdata_version_t bjdata_version) + { + std::vector stack; + write_ubjson_value_or_push(root, add_prefix, use_count, use_type, use_bjdata, bjdata_version, stack); + + while (!stack.empty()) + { + const ubjson_frame current = stack.back(); + const BasicJsonType& j = *current.value; + + if (j.is_array()) + { + const auto& array = *j.m_data.m_value.array; + if (current.array_it == array.cend()) + { + if (!use_count) + { + oa.write_character(to_char_type(']')); + } + stack.pop_back(); + continue; + } + + const BasicJsonType* child = &(*current.array_it); + const bool child_prefix = current.prefix_required; + ++stack.back().array_it; + write_ubjson_value_or_push(*child, child_prefix, use_count, use_type, use_bjdata, bjdata_version, stack); + } + else + { + const auto& object = *j.m_data.m_value.object; + if (current.object_it == object.cend()) + { + if (!use_count) + { + oa.write_character(to_char_type('}')); + } + stack.pop_back(); + continue; + } + + string_t storage; + const string_t& key = sanitize_utf8_for_write(current.object_it->first, j, storage); + write_number_with_ubjson_prefix(key.size(), true, use_bjdata); + oa.write_characters( + reinterpret_cast(key.data()), + key.size()); + const BasicJsonType* child = &(current.object_it->second); + const bool child_prefix = current.prefix_required; + ++stack.back().object_it; + write_ubjson_value_or_push(*child, child_prefix, use_count, use_type, use_bjdata, bjdata_version, stack); + } + } + } + ////////// // BSON // ////////// diff --git a/tests/src/unit-large_json.cpp b/tests/src/unit-large_json.cpp index 40025729a..6ce5626d8 100644 --- a/tests/src/unit-large_json.cpp +++ b/tests/src/unit-large_json.cpp @@ -13,6 +13,7 @@ using nlohmann::json; #include #include +#include #include TEST_CASE("tests on very large JSONs") @@ -354,3 +355,199 @@ TEST_CASE("tests on deeply nested JSONs") } } +namespace +{ +json nested_array(const std::size_t depth, json leaf) +{ + json j = std::move(leaf); + for (std::size_t i = 0; i < depth; ++i) + { + json a = json::array(); + a.push_back(std::move(j)); + j = std::move(a); + } + return j; +} + +json nested_object(const std::size_t depth, json leaf) +{ + json j = std::move(leaf); + for (std::size_t i = 0; i < depth; ++i) + { + json o = json::object(); + o["k"] = std::move(j); + j = std::move(o); + } + return j; +} +} // namespace + +TEST_CASE("issue #5392 - binary writers on deeply nested values") +{ + // 200 is past the point where the writers stop recursing, and still + // shallow enough that from_* and operator== (which still recurse) are fine. + const json deep_array = nested_array(200, json(0)); + const json deep_object = nested_object(200, json("x")); + const json empty_array = nested_array(200, json::array()); + const json empty_object = nested_object(200, json::object()); + const json mixed = nested_object(80, nested_array(80, json(true))); + + SECTION("roundtrip past the recursion bound") + { + CHECK(json::from_cbor(json::to_cbor(deep_array)) == deep_array); + CHECK(json::from_msgpack(json::to_msgpack(deep_array)) == deep_array); + CHECK(json::from_ubjson(json::to_ubjson(deep_array)) == deep_array); + CHECK(json::from_ubjson(json::to_ubjson(deep_array, true, false)) == deep_array); + CHECK(json::from_ubjson(json::to_ubjson(deep_array, true, true)) == deep_array); + CHECK(json::from_bjdata(json::to_bjdata(deep_array)) == deep_array); + + CHECK(json::from_cbor(json::to_cbor(deep_object)) == deep_object); + CHECK(json::from_msgpack(json::to_msgpack(deep_object)) == deep_object); + CHECK(json::from_ubjson(json::to_ubjson(deep_object)) == deep_object); + CHECK(json::from_ubjson(json::to_ubjson(deep_object, true, true)) == deep_object); + CHECK(json::from_bjdata(json::to_bjdata(deep_object)) == deep_object); + + CHECK(json::from_cbor(json::to_cbor(empty_array)) == empty_array); + CHECK(json::from_msgpack(json::to_msgpack(empty_array)) == empty_array); + CHECK(json::from_ubjson(json::to_ubjson(empty_array)) == empty_array); + CHECK(json::from_ubjson(json::to_ubjson(empty_array, true, true)) == empty_array); + + CHECK(json::from_cbor(json::to_cbor(empty_object)) == empty_object); + CHECK(json::from_msgpack(json::to_msgpack(empty_object)) == empty_object); + CHECK(json::from_ubjson(json::to_ubjson(empty_object)) == empty_object); + + CHECK(json::from_cbor(json::to_cbor(mixed)) == mixed); + CHECK(json::from_msgpack(json::to_msgpack(mixed)) == mixed); + CHECK(json::from_ubjson(json::to_ubjson(mixed)) == mixed); + CHECK(json::from_bjdata(json::to_bjdata(mixed)) == mixed); + } + + SECTION("the two ways of writing a value meet at the bound") + { + for (std::size_t depth = 120; depth <= 140; ++depth) + { + CAPTURE(depth); + + const json array = nested_array(depth, json(7)); + CHECK(json::from_cbor(json::to_cbor(array)) == array); + CHECK(json::from_msgpack(json::to_msgpack(array)) == array); + CHECK(json::from_ubjson(json::to_ubjson(array, true, true)) == array); + + const json object = nested_object(depth, json(7)); + CHECK(json::from_cbor(json::to_cbor(object)) == object); + CHECK(json::from_msgpack(json::to_msgpack(object)) == object); + CHECK(json::from_bjdata(json::to_bjdata(object)) == object); + } + } + + SECTION("a BJData ndarray below the bound is still an ndarray") + { + const json ndarray = json({{"_ArrayType_", "uint8"}, {"_ArraySize_", {2, 3}}, {"_ArrayData_", {1, 2, 3, 4, 5, 6}}}); + const json invalid = json({{"_ArrayType_", "nope"}, {"_ArraySize_", {1}}, {"_ArrayData_", {1}}}); + + const json deep_ndarray = nested_array(140, ndarray); + const json deep_invalid = nested_array(140, invalid); + + CHECK(json::from_bjdata(json::to_bjdata(deep_ndarray)) == deep_ndarray); + CHECK(json::from_bjdata(json::to_bjdata(deep_invalid)) == deep_invalid); + CHECK(json::from_bjdata(json::to_bjdata(ndarray)) == ndarray); + } + + SECTION("byte-exact across the switch-over") + { + // nested one-element arrays around the recursion bound: the exact + // bytes a writer produces do not depend on whether it stayed on the + // call stack or moved to the heap one partway through + for (const std::size_t depth : + { + nlohmann::detail::recursion_depth_limit() - 1, nlohmann::detail::recursion_depth_limit(), + nlohmann::detail::recursion_depth_limit() + 1, nlohmann::detail::recursion_depth_limit() + 2 + }) + { + CAPTURE(depth); + const json array = nested_array(depth, json(0)); + + std::vector expected_cbor(depth, 0x81); + expected_cbor.push_back(0x00); + CHECK(json::to_cbor(array) == expected_cbor); + + std::vector expected_msgpack(depth, 0x91); + expected_msgpack.push_back(0x00); + CHECK(json::to_msgpack(array) == expected_msgpack); + + std::string expected_ubjson(depth, '['); + expected_ubjson += "i"; + expected_ubjson += '\0'; + expected_ubjson.append(depth, ']'); + const auto packed_ubjson = json::to_ubjson(array); + CHECK(std::string(packed_ubjson.begin(), packed_ubjson.end()) == expected_ubjson); + } + } + + SECTION("a deep object, and a BJData ndarray, past the recursion bound") + { + const std::size_t depth = nlohmann::detail::recursion_depth_limit() + 50; + + const json object = nested_object(depth, json(42)); + CHECK(json::from_cbor(json::to_cbor(object)) == object); + CHECK(json::from_msgpack(json::to_msgpack(object)) == object); + CHECK(json::from_ubjson(json::to_ubjson(object, true, true)) == object); + CHECK(json::from_bjdata(json::to_bjdata(object)) == object); + + const json ndarray = json({{"_ArrayType_", "uint8"}, {"_ArraySize_", {2, 3}}, {"_ArrayData_", {1, 2, 3, 4, 5, 6}}}); + const json deep_ndarray = nested_array(depth, ndarray); + CHECK(json::from_bjdata(json::to_bjdata(deep_ndarray)) == deep_ndarray); + } + + SECTION("a discarded value past the recursion bound still throws type_error.321") + { + const std::size_t depth = nlohmann::detail::recursion_depth_limit() + 50; + const json discarded_leaf(json::value_t::discarded); + const json deep_discarded = nested_array(depth, discarded_leaf); + + CHECK_THROWS_WITH_AS(json::to_cbor(deep_discarded), "[json.exception.type_error.321] cannot serialize discarded value to CBOR", json::type_error); + CHECK_THROWS_WITH_AS(json::to_msgpack(deep_discarded), "[json.exception.type_error.321] cannot serialize discarded value to MessagePack", json::type_error); + CHECK_THROWS_WITH_AS(json::to_ubjson(deep_discarded), "[json.exception.type_error.321] cannot serialize discarded value to UBJSON", json::type_error); + CHECK_THROWS_WITH_AS(json::to_bjdata(deep_discarded), "[json.exception.type_error.321] cannot serialize discarded value to BJData", json::type_error); + } + + SECTION("does not overflow the C++ stack") + { + const std::size_t depth = 100000; + const json j = json::parse(std::string(depth, '[') + "0" + std::string(depth, ']')); + + std::vector packed; + CHECK_NOTHROW(packed = json::to_cbor(j)); + CHECK(json::from_cbor(packed) == j); + + CHECK_NOTHROW(packed = json::to_msgpack(j)); + CHECK(json::from_msgpack(packed) == j); + + CHECK_NOTHROW(packed = json::to_ubjson(j)); + CHECK(json::from_ubjson(packed) == j); + + CHECK_NOTHROW(packed = json::to_ubjson(j, true, false)); + CHECK(json::from_ubjson(packed) == j); + + CHECK_NOTHROW(packed = json::to_bjdata(j)); + CHECK(json::from_bjdata(packed) == j); + } + + SECTION("regression test for https://issues.oss-fuzz.com/issues/566583014") + { + // 200000 nested one-element CBOR arrays, the innermost holding null; + // round-tripping this used to recurse once per level on the way back + // out through to_cbor(), deep enough to overflow the stack + std::vector v(200000, 0x81); + v.push_back(0xf6); + const json j = json::from_cbor(v); + CHECK(json::to_cbor(j) == v); + + // the MessagePack analogue: fixarray of 1 nesting down to nil + std::vector v_msgpack(200000, 0x91); + v_msgpack.push_back(0xc0); + const json j_msgpack = json::from_msgpack(v_msgpack); + CHECK(json::to_msgpack(j_msgpack) == v_msgpack); + } +} +