diff --git a/.github/workflows/windows.yml b/.github/workflows/windows.yml index 855d8af95..c5220e66c 100644 --- a/.github/workflows/windows.yml +++ b/.github/workflows/windows.yml @@ -166,6 +166,13 @@ jobs: # link, but the binaries clang 11.0.1 and clang 18.1.8 then produce crash # before doctest prints its first line - 39 of 102 tests on clang 18. # Keep the objects small by splitting the test files instead. + # Objects with more than 65535 sections fail to link with "relocation + # truncated to fit: IMAGE_REL_AMD64_REL32 against `.rdata'": the MinGW + # linker stores the section an associative COMDAT section belongs to in + # 16 bits (x_associated in binutils' include/coff/internal.h), so it + # discards the jump tables of inline functions together with the wrong + # function. Each basic_json specialization a test file instantiates adds + # many sections, so test a second one in a file of its own. - name: Run CMake run: cmake -S . -B build ^ -DCMAKE_CXX_COMPILER="C:/Program Files/LLVM/bin/clang++.exe" ^ diff --git a/docs/mkdocs/docs/api/basic_json/binary_t.md b/docs/mkdocs/docs/api/basic_json/binary_t.md index 46b0f7ae2..da13c22a2 100644 --- a/docs/mkdocs/docs/api/basic_json/binary_t.md +++ b/docs/mkdocs/docs/api/basic_json/binary_t.md @@ -64,18 +64,7 @@ values of that type directly to a `basic_json` instance, and they will automatic rather than arrays: ```cpp -using custom_json = nlohmann::basic_json< - nlohmann::ordered_map, // ObjectType - std::vector, // ArrayType - std::string, // StringType - bool, // BooleanType - std::int64_t, // NumberIntegerType - std::uint64_t, // NumberUnsignedType - double, // NumberFloatType - std::allocator, // AllocatorType - nlohmann::adl_serializer, - std::vector // Custom BinaryType ->; +using custom_json = nlohmann::ordered_json::with_binary_t>; std::vector data{std::byte{1}, std::byte{2}, std::byte{3}}; custom_json j = data; // Creates a binary value, not an array diff --git a/docs/mkdocs/docs/api/basic_json/object_t.md b/docs/mkdocs/docs/api/basic_json/object_t.md index 204e5ab28..5e7cab9e9 100644 --- a/docs/mkdocs/docs/api/basic_json/object_t.md +++ b/docs/mkdocs/docs/api/basic_json/object_t.md @@ -26,7 +26,8 @@ To store objects in C++, a type is defined by the template parameters described `StringType` : the type of the keys or names (e.g., `std::string`). The comparison function `std::less` is used to - order elements inside the container. + order elements inside the container. `object_t::key_type` must be implicitly convertible to `string_t` (required by the + binary formats). `AllocatorType` : the allocator to use for objects (e.g., `std::allocator`) diff --git a/docs/mkdocs/docs/community/roadmap.md b/docs/mkdocs/docs/community/roadmap.md index 27afeb4c5..1b1f287a9 100644 --- a/docs/mkdocs/docs/community/roadmap.md +++ b/docs/mkdocs/docs/community/roadmap.md @@ -36,13 +36,26 @@ work items are tracked in the [GitHub milestones](https://github.com/nlohmann/js ## API stability Releases follow [semantic versioning](https://semver.org): a minor or patch release of version 3.x does not break code -that uses the public API. In particular, a 3.x release does not: +that uses the public API, unless that code opts in to a change with a macro as described [below](#version-40). In +particular, a 3.x release does not: -- change the signature of a function (its parameter types, return type, number of parameters, or the const-ness of a - member function); -- remove or rename a function or class; +- make breaking changes to the signature of a function: the types or order of its existing parameters, its return type, + its `noexcept` or `constexpr` specifier, or the const-ness of a member function. New parameters may be added if they + have a default value; +- remove or rename a function or class, or change the template parameters of a public class template; - change which exceptions a function throws, or the [exception ids](../home/exceptions.md); -- change access specifiers or default arguments. +- change access specifiers, or change or remove existing default arguments. New default arguments may be added; +- change the JSON type that a valid input parses to, or the text that `dump()` produces for a valid value; +- accept input that was rejected before, or reject input that was accepted before; +- change the order in which the keys of an object are iterated. The default type sorts keys, and + [`ordered_json`](../api/ordered_json.md) keeps insertion order; +- change when iterators, pointers, or references are invalidated, or the state of a moved-from `basic_json`; +- add or remove implicit conversions from `basic_json`; +- change how `to_json` and `from_json` functions are found, or the behavior of + [`adl_serializer`](../api/adl_serializer/index.md); +- add pure virtual functions to the [`json_sax`](../api/json_sax/index.md) interface; +- remove, rename, renumber, or add enumerators of `value_t`; +- remove or rename a documented macro, CMake option, CMake target, or header, or change what a documented macro does. Exceptions to these rules, for instance when fixing a bug requires changing the exception a function throws, are documented in the [release notes](../home/releases.md). @@ -51,13 +64,14 @@ The following are **not** part of the public API and may change in any release, - The text of exception messages returned by `what()`. Use the [exception id](../home/exceptions.md) to tell errors apart. -- The ABI, including `sizeof(basic_json)` and the memory layout of its values. Recompile your code when you upgrade the - library. The [versioned inline namespace](../features/namespace.md) turns mixing versions into a link error. +- The ABI, including `sizeof(basic_json)` and the memory layout of its values. The + [versioned inline namespace](../features/namespace.md) turns mixing versions into a link error. +- The hash values returned by `std::hash` for `basic_json`. Numbers that compare equal still hash equally. - Everything in namespace `nlohmann::detail`, and macros and type traits that are not documented in the [API reference](../api/basic_json/index.md). -Changes that would break the public API are only added behind a macro whose default keeps the 3.x behavior, see -[Version 4.0](#version-40). +Breaking changes are only added behind a macro whose default keeps the 3.x behavior. See [Version 4.0](#version-40) and +the [macro overview](../features/macros.md). ## Version 4.0 diff --git a/docs/mkdocs/docs/examples/as_base_class.cpp b/docs/mkdocs/docs/examples/as_base_class.cpp index 48357b045..b5b62d78f 100644 --- a/docs/mkdocs/docs/examples/as_base_class.cpp +++ b/docs/mkdocs/docs/examples/as_base_class.cpp @@ -15,19 +15,7 @@ class base_class_with_hidden_members } }; -using json = nlohmann::basic_json < - std::map, - std::vector, - std::string, - bool, - std::int64_t, - std::uint64_t, - double, - std::allocator, - nlohmann::adl_serializer, - std::vector, - base_class_with_hidden_members - >; +using json = nlohmann::json::with_base_class_t; int main() { diff --git a/docs/mkdocs/docs/examples/contains__json_pointer.cpp b/docs/mkdocs/docs/examples/contains__json_pointer.cpp index 14d8514b4..985ad9ec5 100644 --- a/docs/mkdocs/docs/examples/contains__json_pointer.cpp +++ b/docs/mkdocs/docs/examples/contains__json_pointer.cpp @@ -19,25 +19,9 @@ int main() << j.contains("/array/1"_json_pointer) << '\n' << j.contains("/array/-"_json_pointer) << '\n' << j.contains("/array/4"_json_pointer) << '\n' - << j.contains("/baz"_json_pointer) << std::endl; - - try - { - // try to use an array index with leading '0' - j.contains("/array/01"_json_pointer); - } - catch (const json::parse_error& e) - { - std::cout << e.what() << '\n'; - } - - try - { - // try to use an array index that is not a number - j.contains("/array/one"_json_pointer); - } - catch (const json::parse_error& e) - { - std::cout << e.what() << '\n'; - } + << j.contains("/baz"_json_pointer) << '\n' + // an array index with a leading '0' is not found + << j.contains("/array/01"_json_pointer) << '\n' + // an array index that is not a number is not found + << j.contains("/array/one"_json_pointer) << std::endl; } diff --git a/docs/mkdocs/docs/examples/contains__json_pointer.output b/docs/mkdocs/docs/examples/contains__json_pointer.output index dd1eb38c1..b989a7c30 100644 --- a/docs/mkdocs/docs/examples/contains__json_pointer.output +++ b/docs/mkdocs/docs/examples/contains__json_pointer.output @@ -5,3 +5,5 @@ true false false false +false +false diff --git a/docs/mkdocs/docs/examples/parse_error.cpp b/docs/mkdocs/docs/examples/parse_error.cpp index ce15ebe22..2f73ac65f 100644 --- a/docs/mkdocs/docs/examples/parse_error.cpp +++ b/docs/mkdocs/docs/examples/parse_error.cpp @@ -8,7 +8,7 @@ int main() try { // parsing input with a syntax error - json::parse("[1,2,3,]"); + json j = json::parse("[1,2,3,]"); } catch (const json::parse_error& e) { diff --git a/include/nlohmann/detail/json_pointer.hpp b/include/nlohmann/detail/json_pointer.hpp index bf977e80b..02dfeec16 100644 --- a/include/nlohmann/detail/json_pointer.hpp +++ b/include/nlohmann/detail/json_pointer.hpp @@ -16,8 +16,8 @@ #include // ostream #endif // JSON_NO_IO #include // max +#include // map #include // accumulate -#include // set #include // string #include // move #include // vector @@ -366,33 +366,82 @@ class json_pointer private: /*! - @brief the reference token sequences that denote arrays + @brief the pointer prefixes of a flattened object, and which of them denote arrays @ref unflatten collects the pointer prefixes that have a reference token 0 among their children; @ref get_and_create creates arrays exactly below those prefixes and objects everywhere else. Deciding this up front keeps the result independent of the order in which the flattened object is iterated, which is unspecified for some object types. + + The prefixes form a tree and are numbered, so each of them is stored only + once (as a node) rather than as a copy of all of its reference tokens. */ - using array_parents_t = std::set>; + struct prefix_tree + { + // children[id] maps a reference token to the number of the prefix + // extended by that token; number 0 is the empty prefix + std::vector> children; + // is_array[id] is true iff some flattened key has the reference token + // 0 directly below the prefix with number id + std::vector is_array; + + // start with the empty prefix only + prefix_tree() + : children(1) + , is_array(1, false) + {} + + // return the number of the prefix with number id extended by + // reference_token, adding it if it is new + std::size_t add_child(std::size_t id, string_t&& reference_token) + { + if (reference_token == "0") + { + is_array[id] = true; + } + + // read the number before the emplace_back below, which may + // reallocate children and invalidate the iterator + const std::size_t next = children.size(); + const auto inserted = children[id].emplace(std::move(reference_token), next); + const std::size_t child = inserted.first->second; + if (inserted.second) + { + children.emplace_back(); + is_array.push_back(false); + } + return child; + } + + // return the number of the prefix with number id extended by + // reference_token, which must have been added before + std::size_t find_child(std::size_t id, const string_t& reference_token) const + { + const auto it = children[id].find(reference_token); + JSON_ASSERT(it != children[id].end()); + return it->second; + } + }; /*! @brief create and return a reference to the pointed to value - Complexity: Linear in the number of reference tokens. + Complexity: Linear in the number of reference tokens (times the logarithm + of the number of siblings for the prefix lookup). @throw parse_error.106 if an array index begins with '0' @throw parse_error.109 if array index is not a number @throw type_error.313 if value cannot be unflattened */ template - BasicJsonType& get_and_create(BasicJsonType& j, const array_parents_t& array_parents) const + BasicJsonType& get_and_create(BasicJsonType& j, const prefix_tree& tree) const { auto* result = &j; - // the reference tokens that have been consumed so far; used to look up - // whether the value to be created below is an array or an object - std::vector prefix; + // the number of the prefix consumed so far; used to look up whether + // the value to be created below is an array or an object + std::size_t id = 0; // in case no reference tokens exist, return a reference to the JSON value // j which will be overwritten by a primitive value @@ -402,7 +451,7 @@ class json_pointer { case detail::value_t::null: { - if (array_parents.find(prefix) != array_parents.end()) + if (tree.is_array[id]) { // some reference token below this position is 0, so the // value is an array @@ -447,7 +496,7 @@ class json_pointer JSON_THROW(detail::type_error::create(313, "invalid value to unflatten", &j)); } - prefix.push_back(reference_token); + id = tree.find_child(id, reference_token); } return *result; @@ -885,64 +934,131 @@ class json_pointer @param[in,out] result the result object to insert values to @note Empty objects or arrays are flattened to `null`. + + The value is walked with an explicit stack rather than the call stack, so + arbitrarily deeply nested values can be flattened. + + @sa https://github.com/nlohmann/json/issues/5393 */ template static void flatten(const string_t& reference_string, const BasicJsonType& value, BasicJsonType& result) { - switch (value.type()) + using object_const_iterator = typename BasicJsonType::object_t::const_iterator; + + // an array or object being walked: the container, the array index or + // object iterator of the next child, and the length of the path of the + // container itself + struct frame { - case detail::value_t::array: - { - if (value.m_data.m_value.array->empty()) - { - // flatten empty array as null - result[reference_string] = nullptr; - } - else - { - // iterate array and use index as a reference string - for (std::size_t i = 0; i < value.m_data.m_value.array->size(); ++i) - { - flatten(detail::concat(reference_string, '/', std::to_string(i)), - value.m_data.m_value.array->operator[](i), result); - } - } - break; - } + frame(const BasicJsonType* container_, object_const_iterator member_, const std::size_t path_length_) noexcept + : container(container_), member(std::move(member_)), path_length(path_length_) + {} - case detail::value_t::object: - { - if (value.m_data.m_value.object->empty()) - { - // flatten empty object as null - result[reference_string] = nullptr; - } - else - { - // iterate object and use keys as reference string - for (const auto& element : *value.m_data.m_value.object) - { - flatten(detail::concat(reference_string, '/', detail::escape(element.first)), element.second, result); - } - } - break; - } + const BasicJsonType* container; + std::size_t index = 0; + object_const_iterator member; + std::size_t path_length; + }; - case detail::value_t::null: - case detail::value_t::string: - case detail::value_t::boolean: - case detail::value_t::number_integer: - case detail::value_t::number_unsigned: - case detail::value_t::number_float: - case detail::value_t::binary: - case detail::value_t::discarded: - default: + // The containers being flattened are kept on an explicit stack, and + // every child is flattened completely before the next one, so the + // entries come out in the same order as with a recursive walk. The + // path of the value being flattened is kept in one buffer that grows + // and shrinks with the stack, rather than in a new string per level. + std::vector stack; + string_t path = reference_string; + + // flatten `v`, whose path is `path`: primitives and empty containers + // are added to the result right away; other containers get a frame + const auto enter = [&stack, &path, &result](const BasicJsonType & v) + { + switch (v.type()) { - // add a primitive value with its reference string - result[reference_string] = value; - break; + case detail::value_t::array: + { + if (v.m_data.m_value.array->empty()) + { + // flatten empty array as null + result[path] = nullptr; + } + else + { + stack.emplace_back(&v, object_const_iterator(), path.size()); + } + return; + } + + case detail::value_t::object: + { + if (v.m_data.m_value.object->empty()) + { + // flatten empty object as null + result[path] = nullptr; + } + else + { + stack.emplace_back(&v, v.m_data.m_value.object->begin(), path.size()); + } + return; + } + + case detail::value_t::null: + case detail::value_t::string: + case detail::value_t::boolean: + case detail::value_t::number_integer: + case detail::value_t::number_unsigned: + case detail::value_t::number_float: + case detail::value_t::binary: + case detail::value_t::discarded: + default: + { + // add a primitive value with its reference string + result[path] = v; + return; + } + } + }; + + enter(value); + while (!stack.empty()) + { + // the frame is changed through stack.back(): enter() may push a + // frame, which would invalidate a reference to it + const BasicJsonType* const container = stack.back().container; + + // drop the path of the previous child + path.resize(stack.back().path_length); + + if (container->is_array()) + { + const auto& array = *container->m_data.m_value.array; + const std::size_t i = stack.back().index; + if (i == array.size()) + { + stack.pop_back(); + continue; + } + + // iterate array and use index as a reference string + ++stack.back().index; + detail::concat_into(path, '/', detail::to_string(i)); + enter(array[i]); + } + else + { + const object_const_iterator it = stack.back().member; + if (it == container->m_data.m_value.object->end()) + { + stack.pop_back(); + continue; + } + + // iterate object and use keys as reference string + ++stack.back().member; + detail::concat_into(path, '/', detail::escape(it->first)); + enter(it->second); } } } @@ -970,19 +1086,15 @@ class json_pointer // collect the pointer prefixes that have a reference token 0 among // their children; the values below them are arrays, all others are - // objects (see array_parents_t) - array_parents_t array_parents; + // objects (see prefix_tree) + prefix_tree tree; for (const auto& element : *value.m_data.m_value.object) { json_pointer ptr(element.first); - std::vector prefix; + std::size_t id = 0; for (auto& reference_token : ptr.reference_tokens) { - if (reference_token == "0") - { - array_parents.insert(prefix); - } - prefix.push_back(std::move(reference_token)); + id = tree.add_child(id, std::move(reference_token)); } } @@ -998,7 +1110,7 @@ class json_pointer // that if the JSON pointer is "" (i.e., points to the whole value), // function get_and_create returns a reference to the result itself. // An assignment will then create a primitive value. - json_pointer(element.first).get_and_create(result, array_parents) = element.second; + json_pointer(element.first).get_and_create(result, tree) = element.second; } return result; diff --git a/include/nlohmann/detail/output/binary_writer.hpp b/include/nlohmann/detail/output/binary_writer.hpp index 1cb64f08d..212d2658f 100644 --- a/include/nlohmann/detail/output/binary_writer.hpp +++ b/include/nlohmann/detail/output/binary_writer.hpp @@ -85,6 +85,12 @@ template::value, + const string_t&, string_t >::type; using binary_t = typename BasicJsonType::binary_t; using number_float_t = typename BasicJsonType::number_float_t; @@ -244,16 +250,7 @@ class binary_writer case value_t::string: { - string_t storage; - const string_t& value = sanitize_utf8_for_write(*j.m_data.m_value.string, j, storage); - - // step 1: write control byte and the string length - write_cbor_head(0x60, value.size()); - - // step 2: write the string - oa.write_characters( - reinterpret_cast(value.data()), - value.size()); + write_cbor_string(*j.m_data.m_value.string, j); break; } @@ -316,23 +313,20 @@ class binary_writer case value_t::object: { + static_assert( + std::is_convertible < + typename BasicJsonType::object_t::key_type, + string_t >::value, + "object_t::key_type must be implicitly convertible to string_t"); + // step 1: write control byte and the object size write_cbor_head(0xA0, j.m_data.m_value.object->size()); for (const auto& el : *j.m_data.m_value.object) { - // el.first is checked here, against the object as - // diagnostics context, because write_cbor(el.first) - // converts it to a temporary basic_json that would be - // used as the context instead; for error_handler_t::keep - // and ::replace/::ignore the recursive write_cbor(el.first) - // call below handles the key like any other string, so no - // separate check is needed here for those - if (error_handler == error_handler_t::strict) - { - check_utf8(el.first, j); - } - write_cbor(el.first); + // el.first is written directly (not via a temporary + // basic_json), with the object as diagnostics context + write_cbor_string(el.first, j); write_cbor(el.second, depth + 1); } break; @@ -491,39 +485,7 @@ class binary_writer case value_t::string: { - string_t storage; - const string_t& value = sanitize_utf8_for_write(*j.m_data.m_value.string, j, storage); - - // step 1: write control byte and the string length - const auto N = to_msgpack_length(value.size(), j); - if (N <= 31) - { - // fixstr - write_number(static_cast(0xA0 | N)); - } - else if (N <= (std::numeric_limits::max)()) - { - // str 8 - oa.write_character(to_char_type(0xD9)); - write_number(static_cast(N)); - } - else if (N <= (std::numeric_limits::max)()) - { - // str 16 - oa.write_character(to_char_type(0xDA)); - write_number(static_cast(N)); - } - else - { - // str 32 - oa.write_character(to_char_type(0xDB)); - write_number(static_cast(N)); - } - - // step 2: write the string - oa.write_characters( - reinterpret_cast(value.data()), - value.size()); + write_msgpack_string(*j.m_data.m_value.string, j); break; } @@ -629,19 +591,20 @@ class binary_writer case value_t::object: { + static_assert( + std::is_convertible < + typename BasicJsonType::object_t::key_type, + string_t >::value, + "object_t::key_type must be implicitly convertible to string_t"); + // step 1: write control byte and the object size write_msgpack_object_prefix(j.m_data.m_value.object->size(), j); for (const auto& el : *j.m_data.m_value.object) { - // as in write_cbor, el.first is checked here against the - // object as diagnostics context; the recursive call below - // handles keep/replace/ignore like any other string - if (error_handler == error_handler_t::strict) - { - check_utf8(el.first, j); - } - write_msgpack(el.first); + // as in write_cbor, el.first is written directly with the + // object as diagnostics context + write_msgpack_string(el.first, j); write_msgpack(el.second, depth + 1); } break; @@ -819,8 +782,10 @@ class binary_writer for (const auto& el : *j.m_data.m_value.object) { + // a converted key must outlive the reference returned by sanitize_utf8_for_write + const object_key_string_t key_string = el.first; string_t storage; - const string_t& key = sanitize_utf8_for_write(el.first, j, storage); + const string_t& key = sanitize_utf8_for_write(key_string, j, storage); write_number_with_ubjson_prefix(key.size(), true, use_bjdata); oa.write_characters( reinterpret_cast(key.data()), @@ -1025,13 +990,9 @@ class binary_writer continue; } - // el.first is checked here, against the object as diagnostics - // context, like the matching check in write_cbor's object case - if (error_handler == error_handler_t::strict) - { - check_utf8(current.object_it->first, *current.value); - } - write_cbor(current.object_it->first); + // the key is written directly (not via a temporary basic_json), + // with the object as diagnostics context, as in write_cbor + write_cbor_string(current.object_it->first, *current.value); const BasicJsonType* child = &(current.object_it->second); ++stack.back().object_it; write_cbor_value_or_push(*child, stack); @@ -1106,11 +1067,9 @@ class binary_writer continue; } - if (error_handler == error_handler_t::strict) - { - check_utf8(current.object_it->first, *current.value); - } - write_msgpack(current.object_it->first); + // as in write_cbor_iterative, the key is written directly with + // the object as diagnostics context + write_msgpack_string(current.object_it->first, *current.value); const BasicJsonType* child = &(current.object_it->second); ++stack.back().object_it; write_msgpack_value_or_push(*child, stack); @@ -1366,8 +1325,10 @@ class binary_writer continue; } + // a converted key must outlive the reference returned by sanitize_utf8_for_write + const object_key_string_t key_string = current.object_it->first; string_t storage; - const string_t& key = sanitize_utf8_for_write(current.object_it->first, j, storage); + const string_t& key = sanitize_utf8_for_write(key_string, j, storage); write_number_with_ubjson_prefix(key.size(), true, use_bjdata); oa.write_characters( reinterpret_cast(key.data()), @@ -1988,6 +1949,85 @@ class binary_writer } } + /*! + @brief write a CBOR text string + + @a value is checked or sanitized according to @ref error_handler, with + @a context (the string value itself, or the object a key belongs to) used + as diagnostics context; this avoids converting object keys to a temporary + basic_json just to write them + + @note When object_t::key_type is not string_t, @a value is a temporary + string_t converted from the key, which lives only until the end of + the caller's statement. The reference returned by + @ref sanitize_utf8_for_write may refer to it, so it must not escape + this function. + */ + void write_cbor_string(const string_t& value, const BasicJsonType& context) + { + string_t storage; + const string_t& sanitized = sanitize_utf8_for_write(value, context, storage); + + // step 1: write control byte and the string length + write_cbor_head(0x60, sanitized.size()); + + // step 2: write the string + oa.write_characters( + reinterpret_cast(sanitized.data()), + sanitized.size()); + } + + ///////////// + // MsgPack // + ///////////// + + /*! + @brief write a MessagePack str + + @a value is checked or sanitized according to @ref error_handler, with + @a context used as diagnostics context, as in @ref write_cbor_string + + @note As in @ref write_cbor_string, @a value may be a temporary string_t + converted from a key, so the reference returned by + @ref sanitize_utf8_for_write must not escape this function. + */ + void write_msgpack_string(const string_t& value, const BasicJsonType& context) + { + string_t storage; + const string_t& sanitized = sanitize_utf8_for_write(value, context, storage); + + // step 1: write control byte and the string length + const auto N = to_msgpack_length(sanitized.size(), context); + if (N <= 31) + { + // fixstr + write_number(static_cast(0xA0 | N)); + } + else if (N <= (std::numeric_limits::max)()) + { + // str 8 + oa.write_character(to_char_type(0xD9)); + write_number(static_cast(N)); + } + else if (N <= (std::numeric_limits::max)()) + { + // str 16 + oa.write_character(to_char_type(0xDA)); + write_number(static_cast(N)); + } + else + { + // str 32 + oa.write_character(to_char_type(0xDB)); + write_number(static_cast(N)); + } + + // step 2: write the string + oa.write_characters( + reinterpret_cast(sanitized.data()), + sanitized.size()); + } + //////////// // UBJSON // //////////// @@ -2730,6 +2770,11 @@ class binary_writer itself in every case but a sanitized `replace`/`ignore` one, so @a storage must outlive the returned reference only then. + @a s must be an lvalue that outlives the returned reference. An object key + whose `key_type` is not @ref string_t must therefore first be converted + into a named string_t (see @ref object_key_string_t); the deleted overload + below enforces this at compile time. + @param[in] s the string (value or object key) to write @param[in] context the value @a s belongs to (for diagnostics) @param[out] storage backing storage for a sanitized copy @@ -2759,6 +2804,10 @@ class binary_writer } } + /// deleted: anything but a string_t would bind a temporary that dies before the returned reference is used + template < typename T, enable_if_t < !std::is_same::value, int > = 0 > + const string_t& sanitize_utf8_for_write(const T& /*s*/, const BasicJsonType& /*context*/, string_t& /*storage*/) const = delete; // NOLINT(hicpp-use-equals-delete,modernize-use-equals-delete): a private helper's guard, not part of the interface + /*! @brief write an integer in the shortest encoding diff --git a/include/nlohmann/detail/string_utils.hpp b/include/nlohmann/detail/string_utils.hpp index 4ef748f13..e54a56da6 100644 --- a/include/nlohmann/detail/string_utils.hpp +++ b/include/nlohmann/detail/string_utils.hpp @@ -108,8 +108,8 @@ void encode_utf8(std::uint32_t cp, const Out& out) /////////////////// // UTF-8 decoder states used by decode() below -static constexpr std::uint8_t UTF8_ACCEPT = 0; -static constexpr std::uint8_t UTF8_REJECT = 1; +JSON_INLINE_VARIABLE constexpr std::uint8_t UTF8_ACCEPT = 0; +JSON_INLINE_VARIABLE constexpr std::uint8_t UTF8_REJECT = 1; /*! @brief process a byte of a UTF-8 sequence diff --git a/include/nlohmann/detail/view/document_data.hpp b/include/nlohmann/detail/view/document_data.hpp index c48df2d08..962305b9a 100644 --- a/include/nlohmann/detail/view/document_data.hpp +++ b/include/nlohmann/detail/view/document_data.hpp @@ -78,7 +78,7 @@ struct document_data std::size_t text_cap = 0; std::size_t bytes = 0; ///< memory held by edits }; - std::unique_ptr edits{}; ///< created by the first edit // NOLINT(readability-redundant-member-init) + std::unique_ptr edits; ///< created by the first edit /// one allocation for the header and room for `nodes` nodes; large /// documents get a separate node array instead (so it can be trimmed) @@ -104,7 +104,23 @@ struct document_data } }; - document_data() = default; + /// user-provided so that the class-type members can be initialized in the + /// member initialization list (-Weffc++ asks for it, and old GCC rejects a + /// defaulted constructor whose exception specification differs from the + /// implicit one); they cannot take default member initializers, which old + /// Clang (3.4-3.6) rejects. The other members have default member + /// initializers. noexcept: create() constructs into raw memory and could not + /// release it if this threw (the std::string default constructors do not + /// allocate). + document_data() noexcept + : arena() // NOLINT(readability-redundant-member-init) + , owned() // NOLINT(readability-redundant-member-init) + , owned_image() // NOLINT(readability-redundant-member-init) + , indexes() // NOLINT(readability-redundant-member-init) + , index_slots() // NOLINT(readability-redundant-member-init) + , large_objects() // NOLINT(readability-redundant-member-init) + , edits() // NOLINT(readability-redundant-member-init) + {} document_data(const document_data&) = delete; document_data(document_data&&) = delete; document_data& operator=(const document_data&) = delete; @@ -219,6 +235,13 @@ struct document_data template struct navigation { + /// whether the walk follows edits (the template argument as a runtime + /// condition: `Editable && ...` is a constant condition for MSVC, C4127) + static NLOHMANN_VIEW_ALWAYS_INLINE bool editable() noexcept + { + return false; + } + static NLOHMANN_VIEW_ALWAYS_INLINE const node* first(const document_data& /*d*/, const node* n) noexcept { return n + 1; @@ -238,6 +261,11 @@ struct navigation template<> struct navigation { + static NLOHMANN_VIEW_ALWAYS_INLINE bool editable() noexcept + { + return true; + } + static NLOHMANN_VIEW_ALWAYS_INLINE const node* first(const document_data& d, const node* n) noexcept { return d.first_child_edited(n); diff --git a/include/nlohmann/detail/view/lookup.hpp b/include/nlohmann/detail/view/lookup.hpp index 46e1a5eda..35c3fbdd9 100644 --- a/include/nlohmann/detail/view/lookup.hpp +++ b/include/nlohmann/detail/view/lookup.hpp @@ -92,7 +92,7 @@ template const node* find_member(const document_data& d, const node* object, const char* key, std::size_t n) noexcept { using nav = navigation; - if (NLOHMANN_VIEW_UNLIKELY(object->extra != 0) && (!Editable || (object->flags & node_flags::moved) == 0)) + if (NLOHMANN_VIEW_UNLIKELY(object->extra != 0) && (!nav::editable() || (object->flags & node_flags::moved) == 0)) { return find_indexed(d, object, key, n); // a large object (whose members have not been edited) } @@ -134,7 +134,7 @@ template SizeType to_index(IntegerType idx) noexcept { const IntegerType zero = 0; - const auto result = static_cast(idx); // NOLINT(bugprone-signed-char-misuse,cert-str34-c): idx is an index, not a character + const auto result = static_cast(idx); // NOLINT(bugprone-signed-char-misuse,cert-str34-c): negative values are mapped below return (idx < zero || static_cast(result) != idx) ? (std::numeric_limits::max)() : result; } @@ -143,8 +143,9 @@ SizeType to_index(IntegerType idx) noexcept template const node* element_at(const document_data& d, const node* array, std::size_t idx) noexcept { - const node* e = navigation::first(d, array); - if (Editable && (array->flags & node_flags::moved) != 0 && d.edits->moved_cap[array->off] != 0) + using nav = navigation; + const node* e = nav::first(d, array); + if (nav::editable() && (array->flags & node_flags::moved) != 0 && d.edits->moved_cap[array->off] != 0) { return e + idx; // a growable block: one link per element } diff --git a/include/nlohmann/detail/view/macro_scope.hpp b/include/nlohmann/detail/view/macro_scope.hpp index 8bbae1970..e4b685094 100644 --- a/include/nlohmann/detail/view/macro_scope.hpp +++ b/include/nlohmann/detail/view/macro_scope.hpp @@ -28,7 +28,10 @@ #elif defined(_MSC_VER) #define NLOHMANN_VIEW_LIKELY(x) (x) #define NLOHMANN_VIEW_UNLIKELY(x) (x) - #define NLOHMANN_VIEW_ALWAYS_INLINE __forceinline + // plain inline: __forceinline makes MSVC report C4714 (not inlined) for + // function templates it cannot inline, which is an error under /WX; the + // forced inlining is only a performance hint + #define NLOHMANN_VIEW_ALWAYS_INLINE inline #define NLOHMANN_VIEW_NOINLINE __declspec(noinline) #else #define NLOHMANN_VIEW_LIKELY(x) (x) diff --git a/include/nlohmann/json_view.hpp b/include/nlohmann/json_view.hpp index fa2e6bd4c..d7b33a8fc 100644 --- a/include/nlohmann/json_view.hpp +++ b/include/nlohmann/json_view.hpp @@ -37,13 +37,13 @@ #include // ostream #endif #include // string -#include // tuple_element, tuple_size +#include // tuple_element, tuple_size // IWYU pragma: keep #include // decay, enable_if, integral_constant, is_arithmetic, is_base_of, is_integral, is_same, remove_cv, remove_extent #include // unordered_map #include // forward, move #include // vector -#include +#include // IWYU pragma: export // the view builds on internals of the library: both must be the same version #if NLOHMANN_JSON_VERSION_MAJOR != 3 || NLOHMANN_JSON_VERSION_MINOR != 12 || NLOHMANN_JSON_VERSION_PATCH != 0 @@ -71,9 +71,6 @@ NLOHMANN_JSON_NAMESPACE_BEGIN -template -class basic_json_document; - /*! @brief read-only handle to one value of a basic_json_document @@ -1489,7 +1486,7 @@ using ordered_json_editable_view = basic_json_view; NLOHMANN_JSON_NAMESPACE_END // tuple protocol for the items of basic_json_view::items() (structured bindings) -namespace std // NOLINT(cert-dcl58-cpp) +namespace std // NOLINT(cert-dcl58-cpp,bugprone-std-namespace-modification) { #if defined(__clang__) @@ -1513,6 +1510,6 @@ class tuple_element> // NOLINT(cert } // namespace std -#include +#include // IWYU pragma: keep #endif // INCLUDE_NLOHMANN_JSON_VIEW_HPP_ diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 830536d70..8f4eaf449 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -6428,8 +6428,8 @@ void encode_utf8(std::uint32_t cp, const Out& out) /////////////////// // UTF-8 decoder states used by decode() below -static constexpr std::uint8_t UTF8_ACCEPT = 0; -static constexpr std::uint8_t UTF8_REJECT = 1; +JSON_INLINE_VARIABLE constexpr std::uint8_t UTF8_ACCEPT = 0; +JSON_INLINE_VARIABLE constexpr std::uint8_t UTF8_REJECT = 1; /*! @brief process a byte of a UTF-8 sequence @@ -20458,8 +20458,8 @@ NLOHMANN_JSON_NAMESPACE_END #include // ostream #endif // JSON_NO_IO #include // max +#include // map #include // accumulate -#include // set #include // string #include // move #include // vector @@ -20814,33 +20814,82 @@ class json_pointer private: /*! - @brief the reference token sequences that denote arrays + @brief the pointer prefixes of a flattened object, and which of them denote arrays @ref unflatten collects the pointer prefixes that have a reference token 0 among their children; @ref get_and_create creates arrays exactly below those prefixes and objects everywhere else. Deciding this up front keeps the result independent of the order in which the flattened object is iterated, which is unspecified for some object types. + + The prefixes form a tree and are numbered, so each of them is stored only + once (as a node) rather than as a copy of all of its reference tokens. */ - using array_parents_t = std::set>; + struct prefix_tree + { + // children[id] maps a reference token to the number of the prefix + // extended by that token; number 0 is the empty prefix + std::vector> children; + // is_array[id] is true iff some flattened key has the reference token + // 0 directly below the prefix with number id + std::vector is_array; + + // start with the empty prefix only + prefix_tree() + : children(1) + , is_array(1, false) + {} + + // return the number of the prefix with number id extended by + // reference_token, adding it if it is new + std::size_t add_child(std::size_t id, string_t&& reference_token) + { + if (reference_token == "0") + { + is_array[id] = true; + } + + // read the number before the emplace_back below, which may + // reallocate children and invalidate the iterator + const std::size_t next = children.size(); + const auto inserted = children[id].emplace(std::move(reference_token), next); + const std::size_t child = inserted.first->second; + if (inserted.second) + { + children.emplace_back(); + is_array.push_back(false); + } + return child; + } + + // return the number of the prefix with number id extended by + // reference_token, which must have been added before + std::size_t find_child(std::size_t id, const string_t& reference_token) const + { + const auto it = children[id].find(reference_token); + JSON_ASSERT(it != children[id].end()); + return it->second; + } + }; /*! @brief create and return a reference to the pointed to value - Complexity: Linear in the number of reference tokens. + Complexity: Linear in the number of reference tokens (times the logarithm + of the number of siblings for the prefix lookup). @throw parse_error.106 if an array index begins with '0' @throw parse_error.109 if array index is not a number @throw type_error.313 if value cannot be unflattened */ template - BasicJsonType& get_and_create(BasicJsonType& j, const array_parents_t& array_parents) const + BasicJsonType& get_and_create(BasicJsonType& j, const prefix_tree& tree) const { auto* result = &j; - // the reference tokens that have been consumed so far; used to look up - // whether the value to be created below is an array or an object - std::vector prefix; + // the number of the prefix consumed so far; used to look up whether + // the value to be created below is an array or an object + std::size_t id = 0; // in case no reference tokens exist, return a reference to the JSON value // j which will be overwritten by a primitive value @@ -20850,7 +20899,7 @@ class json_pointer { case detail::value_t::null: { - if (array_parents.find(prefix) != array_parents.end()) + if (tree.is_array[id]) { // some reference token below this position is 0, so the // value is an array @@ -20895,7 +20944,7 @@ class json_pointer JSON_THROW(detail::type_error::create(313, "invalid value to unflatten", &j)); } - prefix.push_back(reference_token); + id = tree.find_child(id, reference_token); } return *result; @@ -21333,64 +21382,131 @@ class json_pointer @param[in,out] result the result object to insert values to @note Empty objects or arrays are flattened to `null`. + + The value is walked with an explicit stack rather than the call stack, so + arbitrarily deeply nested values can be flattened. + + @sa https://github.com/nlohmann/json/issues/5393 */ template static void flatten(const string_t& reference_string, const BasicJsonType& value, BasicJsonType& result) { - switch (value.type()) + using object_const_iterator = typename BasicJsonType::object_t::const_iterator; + + // an array or object being walked: the container, the array index or + // object iterator of the next child, and the length of the path of the + // container itself + struct frame { - case detail::value_t::array: - { - if (value.m_data.m_value.array->empty()) - { - // flatten empty array as null - result[reference_string] = nullptr; - } - else - { - // iterate array and use index as a reference string - for (std::size_t i = 0; i < value.m_data.m_value.array->size(); ++i) - { - flatten(detail::concat(reference_string, '/', std::to_string(i)), - value.m_data.m_value.array->operator[](i), result); - } - } - break; - } + frame(const BasicJsonType* container_, object_const_iterator member_, const std::size_t path_length_) noexcept + : container(container_), member(std::move(member_)), path_length(path_length_) + {} - case detail::value_t::object: - { - if (value.m_data.m_value.object->empty()) - { - // flatten empty object as null - result[reference_string] = nullptr; - } - else - { - // iterate object and use keys as reference string - for (const auto& element : *value.m_data.m_value.object) - { - flatten(detail::concat(reference_string, '/', detail::escape(element.first)), element.second, result); - } - } - break; - } + const BasicJsonType* container; + std::size_t index = 0; + object_const_iterator member; + std::size_t path_length; + }; - case detail::value_t::null: - case detail::value_t::string: - case detail::value_t::boolean: - case detail::value_t::number_integer: - case detail::value_t::number_unsigned: - case detail::value_t::number_float: - case detail::value_t::binary: - case detail::value_t::discarded: - default: + // The containers being flattened are kept on an explicit stack, and + // every child is flattened completely before the next one, so the + // entries come out in the same order as with a recursive walk. The + // path of the value being flattened is kept in one buffer that grows + // and shrinks with the stack, rather than in a new string per level. + std::vector stack; + string_t path = reference_string; + + // flatten `v`, whose path is `path`: primitives and empty containers + // are added to the result right away; other containers get a frame + const auto enter = [&stack, &path, &result](const BasicJsonType & v) + { + switch (v.type()) { - // add a primitive value with its reference string - result[reference_string] = value; - break; + case detail::value_t::array: + { + if (v.m_data.m_value.array->empty()) + { + // flatten empty array as null + result[path] = nullptr; + } + else + { + stack.emplace_back(&v, object_const_iterator(), path.size()); + } + return; + } + + case detail::value_t::object: + { + if (v.m_data.m_value.object->empty()) + { + // flatten empty object as null + result[path] = nullptr; + } + else + { + stack.emplace_back(&v, v.m_data.m_value.object->begin(), path.size()); + } + return; + } + + case detail::value_t::null: + case detail::value_t::string: + case detail::value_t::boolean: + case detail::value_t::number_integer: + case detail::value_t::number_unsigned: + case detail::value_t::number_float: + case detail::value_t::binary: + case detail::value_t::discarded: + default: + { + // add a primitive value with its reference string + result[path] = v; + return; + } + } + }; + + enter(value); + while (!stack.empty()) + { + // the frame is changed through stack.back(): enter() may push a + // frame, which would invalidate a reference to it + const BasicJsonType* const container = stack.back().container; + + // drop the path of the previous child + path.resize(stack.back().path_length); + + if (container->is_array()) + { + const auto& array = *container->m_data.m_value.array; + const std::size_t i = stack.back().index; + if (i == array.size()) + { + stack.pop_back(); + continue; + } + + // iterate array and use index as a reference string + ++stack.back().index; + detail::concat_into(path, '/', detail::to_string(i)); + enter(array[i]); + } + else + { + const object_const_iterator it = stack.back().member; + if (it == container->m_data.m_value.object->end()) + { + stack.pop_back(); + continue; + } + + // iterate object and use keys as reference string + ++stack.back().member; + detail::concat_into(path, '/', detail::escape(it->first)); + enter(it->second); } } } @@ -21418,19 +21534,15 @@ class json_pointer // collect the pointer prefixes that have a reference token 0 among // their children; the values below them are arrays, all others are - // objects (see array_parents_t) - array_parents_t array_parents; + // objects (see prefix_tree) + prefix_tree tree; for (const auto& element : *value.m_data.m_value.object) { json_pointer ptr(element.first); - std::vector prefix; + std::size_t id = 0; for (auto& reference_token : ptr.reference_tokens) { - if (reference_token == "0") - { - array_parents.insert(prefix); - } - prefix.push_back(std::move(reference_token)); + id = tree.add_child(id, std::move(reference_token)); } } @@ -21446,7 +21558,7 @@ class json_pointer // that if the JSON pointer is "" (i.e., points to the whole value), // function get_and_create returns a reference to the result itself. // An assignment will then create a primitive value. - json_pointer(element.first).get_and_create(result, array_parents) = element.second; + json_pointer(element.first).get_and_create(result, tree) = element.second; } return result; @@ -22141,6 +22253,12 @@ template::value, + const string_t&, string_t >::type; using binary_t = typename BasicJsonType::binary_t; using number_float_t = typename BasicJsonType::number_float_t; @@ -22300,16 +22418,7 @@ class binary_writer case value_t::string: { - string_t storage; - const string_t& value = sanitize_utf8_for_write(*j.m_data.m_value.string, j, storage); - - // step 1: write control byte and the string length - write_cbor_head(0x60, value.size()); - - // step 2: write the string - oa.write_characters( - reinterpret_cast(value.data()), - value.size()); + write_cbor_string(*j.m_data.m_value.string, j); break; } @@ -22372,23 +22481,20 @@ class binary_writer case value_t::object: { + static_assert( + std::is_convertible < + typename BasicJsonType::object_t::key_type, + string_t >::value, + "object_t::key_type must be implicitly convertible to string_t"); + // step 1: write control byte and the object size write_cbor_head(0xA0, j.m_data.m_value.object->size()); for (const auto& el : *j.m_data.m_value.object) { - // el.first is checked here, against the object as - // diagnostics context, because write_cbor(el.first) - // converts it to a temporary basic_json that would be - // used as the context instead; for error_handler_t::keep - // and ::replace/::ignore the recursive write_cbor(el.first) - // call below handles the key like any other string, so no - // separate check is needed here for those - if (error_handler == error_handler_t::strict) - { - check_utf8(el.first, j); - } - write_cbor(el.first); + // el.first is written directly (not via a temporary + // basic_json), with the object as diagnostics context + write_cbor_string(el.first, j); write_cbor(el.second, depth + 1); } break; @@ -22547,39 +22653,7 @@ class binary_writer case value_t::string: { - string_t storage; - const string_t& value = sanitize_utf8_for_write(*j.m_data.m_value.string, j, storage); - - // step 1: write control byte and the string length - const auto N = to_msgpack_length(value.size(), j); - if (N <= 31) - { - // fixstr - write_number(static_cast(0xA0 | N)); - } - else if (N <= (std::numeric_limits::max)()) - { - // str 8 - oa.write_character(to_char_type(0xD9)); - write_number(static_cast(N)); - } - else if (N <= (std::numeric_limits::max)()) - { - // str 16 - oa.write_character(to_char_type(0xDA)); - write_number(static_cast(N)); - } - else - { - // str 32 - oa.write_character(to_char_type(0xDB)); - write_number(static_cast(N)); - } - - // step 2: write the string - oa.write_characters( - reinterpret_cast(value.data()), - value.size()); + write_msgpack_string(*j.m_data.m_value.string, j); break; } @@ -22685,19 +22759,20 @@ class binary_writer case value_t::object: { + static_assert( + std::is_convertible < + typename BasicJsonType::object_t::key_type, + string_t >::value, + "object_t::key_type must be implicitly convertible to string_t"); + // step 1: write control byte and the object size write_msgpack_object_prefix(j.m_data.m_value.object->size(), j); for (const auto& el : *j.m_data.m_value.object) { - // as in write_cbor, el.first is checked here against the - // object as diagnostics context; the recursive call below - // handles keep/replace/ignore like any other string - if (error_handler == error_handler_t::strict) - { - check_utf8(el.first, j); - } - write_msgpack(el.first); + // as in write_cbor, el.first is written directly with the + // object as diagnostics context + write_msgpack_string(el.first, j); write_msgpack(el.second, depth + 1); } break; @@ -22875,8 +22950,10 @@ class binary_writer for (const auto& el : *j.m_data.m_value.object) { + // a converted key must outlive the reference returned by sanitize_utf8_for_write + const object_key_string_t key_string = el.first; string_t storage; - const string_t& key = sanitize_utf8_for_write(el.first, j, storage); + const string_t& key = sanitize_utf8_for_write(key_string, j, storage); write_number_with_ubjson_prefix(key.size(), true, use_bjdata); oa.write_characters( reinterpret_cast(key.data()), @@ -23081,13 +23158,9 @@ class binary_writer continue; } - // el.first is checked here, against the object as diagnostics - // context, like the matching check in write_cbor's object case - if (error_handler == error_handler_t::strict) - { - check_utf8(current.object_it->first, *current.value); - } - write_cbor(current.object_it->first); + // the key is written directly (not via a temporary basic_json), + // with the object as diagnostics context, as in write_cbor + write_cbor_string(current.object_it->first, *current.value); const BasicJsonType* child = &(current.object_it->second); ++stack.back().object_it; write_cbor_value_or_push(*child, stack); @@ -23162,11 +23235,9 @@ class binary_writer continue; } - if (error_handler == error_handler_t::strict) - { - check_utf8(current.object_it->first, *current.value); - } - write_msgpack(current.object_it->first); + // as in write_cbor_iterative, the key is written directly with + // the object as diagnostics context + write_msgpack_string(current.object_it->first, *current.value); const BasicJsonType* child = &(current.object_it->second); ++stack.back().object_it; write_msgpack_value_or_push(*child, stack); @@ -23422,8 +23493,10 @@ class binary_writer continue; } + // a converted key must outlive the reference returned by sanitize_utf8_for_write + const object_key_string_t key_string = current.object_it->first; string_t storage; - const string_t& key = sanitize_utf8_for_write(current.object_it->first, j, storage); + const string_t& key = sanitize_utf8_for_write(key_string, j, storage); write_number_with_ubjson_prefix(key.size(), true, use_bjdata); oa.write_characters( reinterpret_cast(key.data()), @@ -24044,6 +24117,85 @@ class binary_writer } } + /*! + @brief write a CBOR text string + + @a value is checked or sanitized according to @ref error_handler, with + @a context (the string value itself, or the object a key belongs to) used + as diagnostics context; this avoids converting object keys to a temporary + basic_json just to write them + + @note When object_t::key_type is not string_t, @a value is a temporary + string_t converted from the key, which lives only until the end of + the caller's statement. The reference returned by + @ref sanitize_utf8_for_write may refer to it, so it must not escape + this function. + */ + void write_cbor_string(const string_t& value, const BasicJsonType& context) + { + string_t storage; + const string_t& sanitized = sanitize_utf8_for_write(value, context, storage); + + // step 1: write control byte and the string length + write_cbor_head(0x60, sanitized.size()); + + // step 2: write the string + oa.write_characters( + reinterpret_cast(sanitized.data()), + sanitized.size()); + } + + ///////////// + // MsgPack // + ///////////// + + /*! + @brief write a MessagePack str + + @a value is checked or sanitized according to @ref error_handler, with + @a context used as diagnostics context, as in @ref write_cbor_string + + @note As in @ref write_cbor_string, @a value may be a temporary string_t + converted from a key, so the reference returned by + @ref sanitize_utf8_for_write must not escape this function. + */ + void write_msgpack_string(const string_t& value, const BasicJsonType& context) + { + string_t storage; + const string_t& sanitized = sanitize_utf8_for_write(value, context, storage); + + // step 1: write control byte and the string length + const auto N = to_msgpack_length(sanitized.size(), context); + if (N <= 31) + { + // fixstr + write_number(static_cast(0xA0 | N)); + } + else if (N <= (std::numeric_limits::max)()) + { + // str 8 + oa.write_character(to_char_type(0xD9)); + write_number(static_cast(N)); + } + else if (N <= (std::numeric_limits::max)()) + { + // str 16 + oa.write_character(to_char_type(0xDA)); + write_number(static_cast(N)); + } + else + { + // str 32 + oa.write_character(to_char_type(0xDB)); + write_number(static_cast(N)); + } + + // step 2: write the string + oa.write_characters( + reinterpret_cast(sanitized.data()), + sanitized.size()); + } + //////////// // UBJSON // //////////// @@ -24786,6 +24938,11 @@ class binary_writer itself in every case but a sanitized `replace`/`ignore` one, so @a storage must outlive the returned reference only then. + @a s must be an lvalue that outlives the returned reference. An object key + whose `key_type` is not @ref string_t must therefore first be converted + into a named string_t (see @ref object_key_string_t); the deleted overload + below enforces this at compile time. + @param[in] s the string (value or object key) to write @param[in] context the value @a s belongs to (for diagnostics) @param[out] storage backing storage for a sanitized copy @@ -24815,6 +24972,10 @@ class binary_writer } } + /// deleted: anything but a string_t would bind a temporary that dies before the returned reference is used + template < typename T, enable_if_t < !std::is_same::value, int > = 0 > + const string_t& sanitize_utf8_for_write(const T& /*s*/, const BasicJsonType& /*context*/, string_t& /*storage*/) const = delete; // NOLINT(hicpp-use-equals-delete,modernize-use-equals-delete): a private helper's guard, not part of the interface + /*! @brief write an integer in the shortest encoding diff --git a/single_include/nlohmann/json_view.hpp b/single_include/nlohmann/json_view.hpp index 51f856717..61d930469 100644 --- a/single_include/nlohmann/json_view.hpp +++ b/single_include/nlohmann/json_view.hpp @@ -37,13 +37,13 @@ #include // ostream #endif #include // string -#include // tuple_element, tuple_size +#include // tuple_element, tuple_size // IWYU pragma: keep #include // decay, enable_if, integral_constant, is_arithmetic, is_base_of, is_integral, is_same, remove_cv, remove_extent #include // unordered_map #include // forward, move #include // vector -#include +#include // IWYU pragma: export // the view builds on internals of the library: both must be the same version #if NLOHMANN_JSON_VERSION_MAJOR != 3 || NLOHMANN_JSON_VERSION_MINOR != 12 || NLOHMANN_JSON_VERSION_PATCH != 0 @@ -127,7 +127,10 @@ #elif defined(_MSC_VER) #define NLOHMANN_VIEW_LIKELY(x) (x) #define NLOHMANN_VIEW_UNLIKELY(x) (x) - #define NLOHMANN_VIEW_ALWAYS_INLINE __forceinline + // plain inline: __forceinline makes MSVC report C4714 (not inlined) for + // function templates it cannot inline, which is an error under /WX; the + // forced inlining is only a performance hint + #define NLOHMANN_VIEW_ALWAYS_INLINE inline #define NLOHMANN_VIEW_NOINLINE __declspec(noinline) #else #define NLOHMANN_VIEW_LIKELY(x) (x) @@ -364,7 +367,7 @@ struct document_data std::size_t text_cap = 0; std::size_t bytes = 0; ///< memory held by edits }; - std::unique_ptr edits{}; ///< created by the first edit // NOLINT(readability-redundant-member-init) + std::unique_ptr edits; ///< created by the first edit /// one allocation for the header and room for `nodes` nodes; large /// documents get a separate node array instead (so it can be trimmed) @@ -390,7 +393,23 @@ struct document_data } }; - document_data() = default; + /// user-provided so that the class-type members can be initialized in the + /// member initialization list (-Weffc++ asks for it, and old GCC rejects a + /// defaulted constructor whose exception specification differs from the + /// implicit one); they cannot take default member initializers, which old + /// Clang (3.4-3.6) rejects. The other members have default member + /// initializers. noexcept: create() constructs into raw memory and could not + /// release it if this threw (the std::string default constructors do not + /// allocate). + document_data() noexcept + : arena() // NOLINT(readability-redundant-member-init) + , owned() // NOLINT(readability-redundant-member-init) + , owned_image() // NOLINT(readability-redundant-member-init) + , indexes() // NOLINT(readability-redundant-member-init) + , index_slots() // NOLINT(readability-redundant-member-init) + , large_objects() // NOLINT(readability-redundant-member-init) + , edits() // NOLINT(readability-redundant-member-init) + {} document_data(const document_data&) = delete; document_data(document_data&&) = delete; document_data& operator=(const document_data&) = delete; @@ -505,6 +524,13 @@ struct document_data template struct navigation { + /// whether the walk follows edits (the template argument as a runtime + /// condition: `Editable && ...` is a constant condition for MSVC, C4127) + static NLOHMANN_VIEW_ALWAYS_INLINE bool editable() noexcept + { + return false; + } + static NLOHMANN_VIEW_ALWAYS_INLINE const node* first(const document_data& /*d*/, const node* n) noexcept { return n + 1; @@ -524,6 +550,11 @@ struct navigation template<> struct navigation { + static NLOHMANN_VIEW_ALWAYS_INLINE bool editable() noexcept + { + return true; + } + static NLOHMANN_VIEW_ALWAYS_INLINE const node* first(const document_data& d, const node* n) noexcept { return d.first_child_edited(n); @@ -3148,7 +3179,7 @@ template const node* find_member(const document_data& d, const node* object, const char* key, std::size_t n) noexcept { using nav = navigation; - if (NLOHMANN_VIEW_UNLIKELY(object->extra != 0) && (!Editable || (object->flags & node_flags::moved) == 0)) + if (NLOHMANN_VIEW_UNLIKELY(object->extra != 0) && (!nav::editable() || (object->flags & node_flags::moved) == 0)) { return find_indexed(d, object, key, n); // a large object (whose members have not been edited) } @@ -3190,7 +3221,7 @@ template SizeType to_index(IntegerType idx) noexcept { const IntegerType zero = 0; - const auto result = static_cast(idx); // NOLINT(bugprone-signed-char-misuse,cert-str34-c): idx is an index, not a character + const auto result = static_cast(idx); // NOLINT(bugprone-signed-char-misuse,cert-str34-c): negative values are mapped below return (idx < zero || static_cast(result) != idx) ? (std::numeric_limits::max)() : result; } @@ -3199,8 +3230,9 @@ SizeType to_index(IntegerType idx) noexcept template const node* element_at(const document_data& d, const node* array, std::size_t idx) noexcept { - const node* e = navigation::first(d, array); - if (Editable && (array->flags & node_flags::moved) != 0 && d.edits->moved_cap[array->off] != 0) + using nav = navigation; + const node* e = nav::first(d, array); + if (nav::editable() && (array->flags & node_flags::moved) != 0 && d.edits->moved_cap[array->off] != 0) { return e + idx; // a growable block: one link per element } @@ -6972,9 +7004,6 @@ NLOHMANN_JSON_NAMESPACE_END NLOHMANN_JSON_NAMESPACE_BEGIN -template -class basic_json_document; - /*! @brief read-only handle to one value of a basic_json_document @@ -8390,7 +8419,7 @@ using ordered_json_editable_view = basic_json_view; NLOHMANN_JSON_NAMESPACE_END // tuple protocol for the items of basic_json_view::items() (structured bindings) -namespace std // NOLINT(cert-dcl58-cpp) +namespace std // NOLINT(cert-dcl58-cpp,bugprone-std-namespace-modification) { #if defined(__clang__) @@ -8443,6 +8472,6 @@ class tuple_element> // NOLINT(cert #undef NLOHMANN_VIEW_SSSE3_TARGET #undef NLOHMANN_VIEW_VECTOR #undef NLOHMANN_VIEW_VECTOR_UTF8 - +// IWYU pragma: keep #endif // INCLUDE_NLOHMANN_JSON_VIEW_HPP_ diff --git a/tests/CMakeLists.txt b/tests/CMakeLists.txt index 959be5434..bbd36bf08 100644 --- a/tests/CMakeLists.txt +++ b/tests/CMakeLists.txt @@ -138,7 +138,8 @@ json_test_set_test_options(test-disabled_exceptions # only the #972 regression test needs thirdparty/fifo_map on its include path json_test_set_test_options(test-regression1 LINK_LIBRARIES fifo_map_include) -# GCC's false -Warray-bounds error with JSON_DIAGNOSTICS only shows up when optimizing (#5742). +# Regression test for GCC's false -Warray-bounds error with JSON_DIAGNOSTICS (#5742, fixed in #5585). It only +# showed up when optimizing, so build this test with -O3 and the warning as an error. # -O3 makes the optimizer-driven warnings of the ci_test_gcc flag set (-Winline, # -Wsuggest-attribute=...) fire on the library's inline functions; they are not # what this test checks, so turn them off for it. diff --git a/tests/src/custom_object_key_type.hpp b/tests/src/custom_object_key_type.hpp new file mode 100644 index 000000000..fa23736d4 --- /dev/null +++ b/tests/src/custom_object_key_type.hpp @@ -0,0 +1,75 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +#include +#include +#include +#include +#include + +namespace custom_object_key_test +{ +class key +{ + public: + key() = default; + + key(const char* value) + : m_value(value) + {} + + key(std::string value) + : m_value(std::move(value)) + {} + + operator std::string() const + { + return m_value; + } + + // Required by JSON_DIAGNOSTICS, which reads object keys through data() + // when building the path of an exception. + const char* data() const noexcept + { + return m_value.data(); + } + + friend bool operator<(const key& lhs, const key& rhs) + { + return lhs.m_value < rhs.m_value; + } + + private: + std::string m_value; +}; + +template +class object + : public std::map < + key, + Value, + std::less, // NOLINT(modernize-use-transparent-functors) + typename std::allocator_traits::template rebind_alloc < + std::pair>> +{ + private: + using allocator_type = + typename std::allocator_traits::template rebind_alloc < + std::pair>; + + using base_type = + std::map, allocator_type>; // NOLINT(modernize-use-transparent-functors) + + public: + using base_type::base_type; +}; + +using json = nlohmann::basic_json; +} // namespace custom_object_key_test diff --git a/tests/src/unit-binary_utf8_error_handler.cpp b/tests/src/unit-binary_utf8_error_handler.cpp index c3f2435fd..114c92885 100644 --- a/tests/src/unit-binary_utf8_error_handler.cpp +++ b/tests/src/unit-binary_utf8_error_handler.cpp @@ -12,7 +12,11 @@ #include using nlohmann::json; +#include +#include +#include #include +#include #include namespace @@ -55,6 +59,53 @@ std::string dump_and_parse(const std::string& raw, eh error_handler) return json::parse(json(raw).dump(-1, ' ', false, error_handler)).get(); } +// an object key type that is not string_t, but converts implicitly to it; +// data() is only used when JSON_DIAGNOSTICS is enabled +DOCTEST_CLANG_SUPPRESS_WARNING_PUSH +DOCTEST_CLANG_SUPPRESS_WARNING("-Wunused-member-function") +class converting_key +{ + public: + converting_key(const char* s) : m_value(s) {} // NOLINT(google-explicit-constructor,hicpp-explicit-conversions) + converting_key(std::string s) : m_value(std::move(s)) {} // NOLINT(google-explicit-constructor,hicpp-explicit-conversions) + + // the conversion yields a temporary string_t + operator std::string() const // NOLINT(google-explicit-constructor,hicpp-explicit-conversions) + { + return m_value; + } + + // read by the exception messages when JSON_DIAGNOSTICS is enabled + const char* data() const noexcept + { + return m_value.data(); + } + + friend bool operator<(const converting_key& lhs, const converting_key& rhs) + { + return lhs.m_value < rhs.m_value; + } + + private: + std::string m_value; +}; +DOCTEST_CLANG_SUPPRESS_WARNING_POP + +// ObjectType using converting_key; the Key template argument is ignored +template +class converting_key_object : public std::map, // NOLINT(modernize-use-transparent-functors) + typename std::allocator_traits::template rebind_alloc>> +{ + using base_type = std::map, // NOLINT(modernize-use-transparent-functors) + typename std::allocator_traits::template rebind_alloc>>; + + public: + using base_type::base_type; + using base_type::operator=; +}; + +using converting_key_json = nlohmann::basic_json; + } // namespace TEST_CASE("UTF-8 error_handler for the binary readers and writers") @@ -370,3 +421,109 @@ TEST_CASE("UTF-8 error_handler for the binary readers and writers") CHECK(json::from_bson(bson_bytes)["k"].get() == ill_formed_cases()[0].bytes); } } + +// The UBJSON and BJData writers bind the (possibly sanitized) key to a const +// string_t&. If key_type is not string_t but converts to it, the converted +// temporary must outlive that reference; this was a use-after-scope found by +// AddressSanitizer. Keys exceed the small string optimization on purpose. +TEST_CASE("UBJSON and BJData writers with an object_t whose key_type is not string_t") +{ + const std::string long_prefix(70, 'k'); + + SECTION("well-formed keys, every error_handler") + { + const std::string key1 = long_prefix + "-first"; + const std::string key2 = long_prefix + "-second"; + + converting_key_json::object_t o; + o.emplace(converting_key(key1), 1); + o.emplace(converting_key(key2), "value"); + const converting_key_json v(std::move(o)); + + json expected; + expected[key1] = 1; + expected[key2] = "value"; + + const std::array, 3> combos = {{{false, false}, {true, false}, {true, true}}}; + for (const auto h : all_handlers()) + { + CAPTURE(static_cast(h)) + for (const auto& combo : combos) + { + const bool use_count = combo.first; + const bool use_type = combo.second; + CAPTURE(use_count) + CAPTURE(use_type) + + CHECK(json::from_ubjson(converting_key_json::to_ubjson(v, use_count, use_type, h)) == expected); + CHECK(json::from_bjdata(converting_key_json::to_bjdata(v, use_count, use_type, json::bjdata_version_t::draft2, h)) == expected); + CHECK(json::from_bjdata(converting_key_json::to_bjdata(v, use_count, use_type, json::bjdata_version_t::draft3, h)) == expected); + } + } + } + + SECTION("ill-formed keys") + { + for (const auto& c : ill_formed_cases()) + { + CAPTURE(c.name) + const std::string key = long_prefix + c.bytes; + + converting_key_json::object_t o; + o.emplace(converting_key(key), 1); + const converting_key_json v(std::move(o)); + + CHECK_THROWS_AS(converting_key_json::to_ubjson(v, false, false, eh::strict), converting_key_json::type_error&); + CHECK_THROWS_AS(converting_key_json::to_bjdata(v, false, false, json::bjdata_version_t::draft2, eh::strict), converting_key_json::type_error&); + + for (const auto h : + { + eh::replace, eh::ignore + }) + { + CAPTURE(static_cast(h)) + const std::string expected = dump_and_parse(key, h); + + CHECK(json::from_ubjson(converting_key_json::to_ubjson(v, false, false, h)).begin().key() == expected); + CHECK(json::from_bjdata(converting_key_json::to_bjdata(v, false, false, json::bjdata_version_t::draft2, h)).begin().key() == expected); + } + + CHECK(json::from_ubjson(converting_key_json::to_ubjson(v, false, false, eh::keep)).begin().key() == key); + CHECK(json::from_bjdata(converting_key_json::to_bjdata(v, false, false, json::bjdata_version_t::draft2, eh::keep)).begin().key() == key); + } + } + + SECTION("nested deeper than the recursion limit") + { + // wrap the previous value, innermost first + converting_key_json v = 42; + json expected = 42; + for (int i = 199; i >= 0; --i) + { + const std::string key = "level-" + std::to_string(i) + "-" + std::string(64, 'x'); + + converting_key_json::object_t o; + o.emplace(converting_key(key), std::move(v)); + v = converting_key_json(std::move(o)); + + json e; + e[key] = std::move(expected); + expected = std::move(e); + } + + for (const auto h : all_handlers()) + { + CAPTURE(static_cast(h)) + for (const bool use_count : + { + false, true + }) + { + CAPTURE(use_count) + + CHECK(json::from_ubjson(converting_key_json::to_ubjson(v, use_count, false, h)) == expected); + CHECK(json::from_bjdata(converting_key_json::to_bjdata(v, use_count, false, json::bjdata_version_t::draft2, h)) == expected); + } + } + } +} diff --git a/tests/src/unit-bson.cpp b/tests/src/unit-bson.cpp index ea5ca3130..03c901cee 100644 --- a/tests/src/unit-bson.cpp +++ b/tests/src/unit-bson.cpp @@ -46,9 +46,7 @@ class huge_binary_t : public std::vector } }; -using huge_binary_json = nlohmann::basic_json < - std::map, std::vector, std::string, bool, std::int64_t, std::uint64_t, - double, std::allocator, nlohmann::adl_serializer, huge_binary_t, void >; +using huge_binary_json = nlohmann::json::with_binary_t; // a string type that can be made to report a size beyond INT32_MAX without // allocating that much memory, so BSON length overflow can be tested for @@ -96,9 +94,7 @@ class huge_string_t : public std::string bool pretend_huge = false; }; -using huge_string_json = nlohmann::basic_json < - std::map, std::vector, huge_string_t, bool, std::int64_t, std::uint64_t, - double, std::allocator, nlohmann::adl_serializer, std::vector, void >; +using huge_string_json = nlohmann::json::with_string_t; } // namespace TEST_CASE("BSON") diff --git a/tests/src/unit-class_lexer.cpp b/tests/src/unit-class_lexer.cpp index e5f82b53c..daa855d33 100644 --- a/tests/src/unit-class_lexer.cpp +++ b/tests/src/unit-class_lexer.cpp @@ -1794,9 +1794,9 @@ TEST_CASE("string scanning kernels") } // eight bytes as a little-endian word, at any alignment - const unsigned char bytes[16] = {0x01, 0x02, 0x03, 0x04, 0x05, 0x06, 0x07, 0x08, 0x09, 0x0A, 0x0B, 0x0C, 0x0D, 0x0E, 0x0F, 0xFF}; - CHECK(nlohmann::detail::read_eight_bytes(bytes) == 0x0807060504030201u); - CHECK(nlohmann::detail::read_eight_bytes(bytes + 1) == 0x0908070605040302u); - CHECK(nlohmann::detail::read_eight_bytes(bytes + 8) == 0xFF0F0E0D0C0B0A09u); - CHECK(nlohmann::detail::read_eight_bytes(reinterpret_cast(bytes) + 3) == 0x0B0A090807060504u); + const std::array bytes = {{0x01, 0x02, 0x03, 0x04, 0x05, 0x06, 0x07, 0x08, 0x09, 0x0A, 0x0B, 0x0C, 0x0D, 0x0E, 0x0F, 0xFF}}; + CHECK(nlohmann::detail::read_eight_bytes(bytes.data()) == 0x0807060504030201u); + CHECK(nlohmann::detail::read_eight_bytes(bytes.data() + 1) == 0x0908070605040302u); + CHECK(nlohmann::detail::read_eight_bytes(bytes.data() + 8) == 0xFF0F0E0D0C0B0A09u); + CHECK(nlohmann::detail::read_eight_bytes(reinterpret_cast(bytes.data()) + 3) == 0x0B0A090807060504u); } diff --git a/tests/src/unit-custom-base-class.cpp b/tests/src/unit-custom-base-class.cpp index 9718d1cda..673c7ddbb 100644 --- a/tests/src/unit-custom-base-class.cpp +++ b/tests/src/unit-custom-base-class.cpp @@ -400,20 +400,7 @@ class base_class_with_hidden_members std::size_t m_size = 42; }; -using json_with_hidden_base_members = - nlohmann::basic_json < - std::map, - std::vector, - std::string, - bool, - std::int64_t, - std::uint64_t, - double, - std::allocator, - nlohmann::adl_serializer, - std::vector, - base_class_with_hidden_members - >; +using json_with_hidden_base_members = nlohmann::json::with_base_class_t; TEST_CASE("JSON Node as_base_class") { @@ -459,19 +446,7 @@ struct const_member_base const int id = 7; // NOLINT(misc-non-private-member-variables-in-classes) }; -using json_with_const_base = nlohmann::basic_json < - std::map, - std::vector, - std::string, - bool, - std::int64_t, - std::uint64_t, - double, - std::allocator, - nlohmann::adl_serializer, - std::vector, - const_member_base - >; +using json_with_const_base = nlohmann::json::with_base_class_t; // build an array nested @a depth levels deep, with the innermost value 1; // every level is constructed (never assigned), since const_member_base does diff --git a/tests/src/unit-custom-binary-type.cpp b/tests/src/unit-custom-binary-type.cpp index d885376cb..6b504bd51 100644 --- a/tests/src/unit-custom-binary-type.cpp +++ b/tests/src/unit-custom-binary-type.cpp @@ -26,15 +26,11 @@ namespace // a BinaryType whose value type is signed: the elements must still be // processed as the numbers 0..255 -using char_binary_json = nlohmann::basic_json < - std::map, std::vector, std::string, bool, std::int64_t, std::uint64_t, - double, std::allocator, nlohmann::adl_serializer, std::vector, void >; +using char_binary_json = nlohmann::json::with_binary_t>; #ifdef JSON_HAS_CPP_17 // a BinaryType whose value type is not an integer type at all - using byte_binary_json = nlohmann::basic_json < - std::map, std::vector, std::string, bool, std::int64_t, std::uint64_t, - double, std::allocator, nlohmann::adl_serializer, std::vector, void >; + using byte_binary_json = nlohmann::json::with_binary_t>; #endif } // namespace diff --git a/tests/src/unit-custom-object-key-type.cpp b/tests/src/unit-custom-object-key-type.cpp new file mode 100644 index 000000000..9c64ebce0 --- /dev/null +++ b/tests/src/unit-custom-object-key-type.cpp @@ -0,0 +1,127 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#include "doctest_compatibility.h" + +#include + +#include +#include +#include + +#include "custom_object_key_type.hpp" + +// These tests instantiate a second basic_json specialization. They live in +// their own file rather than in unit-cbor.cpp and unit-msgpack.cpp to keep +// those objects below 65535 sections: the MinGW linker stores the section a +// COMDAT section is associated with in 16 bits, so it misplaces the jump +// tables of larger objects (see the clang job in windows.yml). + +TEST_CASE("CBOR supports custom object key types") +{ + using custom_json = custom_object_key_test::json; + using custom_key = custom_object_key_test::key; + + custom_json::object_t object; + object.emplace(custom_key{"short"}, 1); + object.emplace( + custom_key{"a key longer than twenty-three characters"}, + 2); + + const custom_json value(std::move(object)); + const auto encoded = custom_json::to_cbor(value); + + CHECK(nlohmann::json::from_cbor(encoded) == nlohmann::json + { + {"short", 1}, + {"a key longer than twenty-three characters", 2} + }); +} + +TEST_CASE("CBOR supports custom object key types nested deeper than the recursion depth limit") +{ + // below detail::recursion_depth_limit(), keys are written by + // write_cbor_iterative instead of write_cbor + using custom_json = custom_object_key_test::json; + using custom_key = custom_object_key_test::key; + + const std::size_t depth = nlohmann::detail::recursion_depth_limit() + 10; + + custom_json value = 1; + nlohmann::json expected = 1; + for (std::size_t i = 0; i < depth; ++i) + { + // alternate short keys with ones long enough to need a length byte + const std::string name = (i % 2 == 0) ? "k" + std::to_string(i) + : "a key longer than thirty-one characters " + std::to_string(i); + + custom_json::object_t object; + object.emplace(custom_key{name}, std::move(value)); + value = custom_json(std::move(object)); + + nlohmann::json::object_t expected_object; + expected_object.emplace(name, std::move(expected)); + expected = nlohmann::json(std::move(expected_object)); + } + + const auto encoded = custom_json::to_cbor(value); + CHECK(encoded == nlohmann::json::to_cbor(expected)); + CHECK(nlohmann::json::from_cbor(encoded) == expected); +} + +TEST_CASE("MessagePack supports custom object key types") +{ + using custom_json = custom_object_key_test::json; + using custom_key = custom_object_key_test::key; + + custom_json::object_t object; + object.emplace(custom_key{"short"}, 1); + object.emplace( + custom_key{"a key longer than thirty-one characters"}, + 2); + + const custom_json value(std::move(object)); + const auto encoded = custom_json::to_msgpack(value); + + CHECK(nlohmann::json::from_msgpack(encoded) == nlohmann::json + { + {"short", 1}, + {"a key longer than thirty-one characters", 2} + }); +} + +TEST_CASE("MessagePack supports custom object key types nested deeper than the recursion depth limit") +{ + // below detail::recursion_depth_limit(), keys are written by + // write_msgpack_iterative instead of write_msgpack + using custom_json = custom_object_key_test::json; + using custom_key = custom_object_key_test::key; + + const std::size_t depth = nlohmann::detail::recursion_depth_limit() + 10; + + custom_json value = 1; + nlohmann::json expected = 1; + for (std::size_t i = 0; i < depth; ++i) + { + // alternate short keys with ones long enough to need a length byte + const std::string name = (i % 2 == 0) ? "k" + std::to_string(i) + : "a key longer than thirty-one characters " + std::to_string(i); + + custom_json::object_t object; + object.emplace(custom_key{name}, std::move(value)); + value = custom_json(std::move(object)); + + nlohmann::json::object_t expected_object; + expected_object.emplace(name, std::move(expected)); + expected = nlohmann::json(std::move(expected_object)); + } + + const auto encoded = custom_json::to_msgpack(value); + CHECK(encoded == nlohmann::json::to_msgpack(expected)); + CHECK(nlohmann::json::from_msgpack(encoded) == expected); +} diff --git a/tests/src/unit-json_pointer.cpp b/tests/src/unit-json_pointer.cpp index b4ed2c9cc..1faf511f4 100644 --- a/tests/src/unit-json_pointer.cpp +++ b/tests/src/unit-json_pointer.cpp @@ -939,3 +939,114 @@ TEST_CASE("unescaping keeps a '~' that does not start an escape sequence") nlohmann::detail::unescape(s); CHECK(s == "~/~"); } + +TEST_CASE("flatten of structured values") +{ + SECTION("values nested too deeply for the call stack (#5393)") + { + // flatten() used to recurse once per nesting level + const std::size_t depth = 100000; + for (const bool objects : + { + false, true + }) + { + CAPTURE(objects) + std::string text; + std::string path; + for (std::size_t i = 0; i < depth; ++i) + { + text += objects ? "{\"a\":" : "["; + path += objects ? "/a" : "/0"; + } + text += "0"; + text += std::string(depth, objects ? '}' : ']'); + const auto value = json::parse(text); + + const auto flat = value.flatten(); + REQUIRE(flat.size() == 1); + REQUIRE(flat.begin().key().size() == path.size()); + CHECK(flat.begin().key() == path); + CHECK(flat.begin().value() == 0); + + // unflatten() is linear in the depth, so the value roundtrips + CHECK(flat.unflatten() == value); + } + } + + SECTION("unflatten of a deeply nested pointer") + { + const std::size_t depth = 100000; + for (const bool objects : + { + false, true + }) + { + CAPTURE(objects) + std::string path; + for (std::size_t i = 0; i < depth; ++i) + { + path += objects ? "/a" : "/0"; + } + + json flat = json::object(); + flat[path] = 1; + const json value = flat.unflatten(); + + // walk down iteratively + std::size_t levels = 0; + const json* current = &value; + while (objects ? current->is_object() : current->is_array()) + { + REQUIRE(current->size() == 1); + current = objects ? ¤t->at("a") : ¤t->at(0); + ++levels; + } + CHECK(levels == depth); + CHECK(*current == 1); + } + } + + SECTION("unflatten does not depend on the iteration order") + { + // the "0" key comes after its sibling in iteration order + const nlohmann::ordered_json flat_array = nlohmann::ordered_json::parse(R"({"/a/1": 2, "/a/0": 1})"); + CHECK(flat_array.unflatten() == nlohmann::ordered_json::parse(R"({"a": [1, 2]})")); + + const nlohmann::ordered_json flat_object = nlohmann::ordered_json::parse(R"({"/b/1": 2})"); + CHECK(flat_object.unflatten() == nlohmann::ordered_json::parse(R"({"b": {"1": 2}})")); + } + + SECTION("objects and arrays interleaved") + { + const json value = + { + {"a", {1, {{"b", json::array()}, {"c", json::object()}}, json::array({{{"x~/", {true, nullptr}}}})}}, + {"a/b", {{"~", 1}}}, + {"z", "s"} + }; + + const json expected = + { + {"/a/0", 1}, + {"/a/1/b", nullptr}, + {"/a/1/c", nullptr}, + {"/a/2/0/x~0~1/0", true}, + {"/a/2/0/x~0~1/1", nullptr}, + {"/a~1b/~0", 1}, + {"/z", "s"} + }; + + CHECK(value.flatten() == expected); + } + + SECTION("order of the entries of an ordered_json") + { + const auto value = nlohmann::ordered_json::parse( + R"({"z":"s","a/b":{"~":1,"k":[]},"a":[1,{"c":{},"b":[]},[{"x~/":[true,null],"w":2}]]})"); + + const auto flat = value.flatten(); + CHECK(flat.dump() == + R"({"/z":"s","/a~1b/~0":1,"/a~1b/k":null,"/a/0":1,"/a/1/c":null,"/a/1/b":null,"/a/2/0/x~0~1/0":true,"/a/2/0/x~0~1/1":null,"/a/2/0/w":2})"); + } +} diff --git a/tests/src/unit-json_view.cpp b/tests/src/unit-json_view.cpp index 77c5d1210..538a4a69b 100644 --- a/tests/src/unit-json_view.cpp +++ b/tests/src/unit-json_view.cpp @@ -66,7 +66,7 @@ struct oversized_input using value_type = char; std::size_t claimed; - const char* data() const + const char* data() const // NOLINT(readability-convert-member-functions-to-static): container interface { return "[1]"; } @@ -363,13 +363,13 @@ TEST_CASE("json_view") CHECK(json_document::parse(std::vector(text.begin(), text.end())).owns_source()); // a const rvalue cannot be moved from, and is not borrowed (it may be a // temporary): it is copied, as is a const rvalue of any container - const std::string const_text = text; + const std::string const_text = text; // NOLINT(performance-unnecessary-copy-initialization) const json_document from_const_rvalue = json_document::parse(std::move(const_text)); // NOLINT(performance-move-const-arg,hicpp-move-const-arg) CHECK(from_const_rvalue.owns_source()); - CHECK(from_const_rvalue.source().data() != const_text.data()); + CHECK(from_const_rvalue.source().data() != const_text.data()); // NOLINT(bugprone-use-after-move,hicpp-invalid-access-moved): const, not moved from CHECK(from_const_rvalue.root().materialize() == expected); json_document read_const_rvalue; - read_const_rvalue.read(std::move(const_text)); // NOLINT(performance-move-const-arg,hicpp-move-const-arg) + read_const_rvalue.read(std::move(const_text)); // NOLINT(performance-move-const-arg,hicpp-move-const-arg,bugprone-use-after-move,hicpp-invalid-access-moved) CHECK(read_const_rvalue.owns_source()); CHECK(read_const_rvalue.root().materialize() == expected); const std::vector const_chars(text.begin(), text.end()); @@ -744,7 +744,7 @@ TEST_CASE("json_view element access and iteration") }) { const std::string key(n, 'k'); - const json_document dk = json_document::parse("{\"" + key + "\":1,\"" + key + "x\":2,\"" + key + "\":3,\"" + key + "\":4}"); + const json_document dk = json_document::parse("{\"" + key + "\":1,\"" + key + "x\":2,\"" + key + "\":3,\"" + key + "\":4}"); // NOLINT(performance-inefficient-string-concatenation) CAPTURE(n) CHECK(dk.root()[key].materialize() == 1); CHECK(dk.root().at(key).materialize() == 1); @@ -861,6 +861,8 @@ TEST_CASE("json_view element access and iteration") const json_document d = json_document::parse("[10,20,30]"); const json_view v = d.root(); const json j = v.materialize(); + // (a variable: a cast of a constant to int32_t is a useless cast to GCC) + const std::int32_t int32_index = 1; // (compile-time: no overload is ambiguous) CHECK(v[0].materialize() == 10); CHECK(v[1].materialize() == 20); @@ -876,7 +878,7 @@ TEST_CASE("json_view element access and iteration") CHECK(v[static_cast(2)].materialize() == 30); CHECK(v[std::int8_t(1)].materialize() == 20); CHECK(v[std::int16_t(2)].materialize() == 30); - CHECK(v[std::int32_t(1)].materialize() == 20); + CHECK(v[int32_index].materialize() == 20); CHECK(v[std::int64_t(2)].materialize() == 30); CHECK(v[std::uint32_t(0)].materialize() == 10); CHECK(v[std::uint64_t(1)].materialize() == 20); @@ -891,7 +893,7 @@ TEST_CASE("json_view element access and iteration") CHECK(v.at(2ULL).materialize() == 30); CHECK(v.at(static_cast(1)).materialize() == 20); CHECK(v.at(static_cast(2)).materialize() == 30); - CHECK(v.at(std::int32_t(0)).materialize() == 10); + CHECK(v.at(int32_index - 1).materialize() == 10); CHECK(v.at(std::uint32_t(0)).materialize() == 10); CHECK(v.at(std::int64_t(0)).materialize() == 10); CHECK(v.at(std::uint64_t(1)).materialize() == 20); @@ -975,8 +977,8 @@ TEST_CASE("json_view element access and iteration") pairs += std::string(key) + "=" + value.materialize().dump() + ";"; } CHECK(pairs == "a=1;b=[true,false];"); - static_assert(std::tuple_size::value == 2, ""); - static_assert(std::is_same::type, json_view>::value, ""); + static_assert(std::tuple_size::value == 2, "tuple_size of an item is 2"); + static_assert(std::is_same::type, json_view>::value, "the second element of an item is a view"); #endif } } @@ -1177,19 +1179,19 @@ TEST_CASE("json_view values") switch (i % 5) // NOLINT(hicpp-multiway-paths-covered) { case 0: - std::snprintf(buf.data(), buf.size(), "%.17g", d); // NOLINT(cppcoreguidelines-pro-type-vararg,hicpp-vararg) + static_cast(std::snprintf(buf.data(), buf.size(), "%.17g", d)); // NOLINT(cppcoreguidelines-pro-type-vararg,hicpp-vararg) break; case 1: - std::snprintf(buf.data(), buf.size(), "%.15g", d); // NOLINT(cppcoreguidelines-pro-type-vararg,hicpp-vararg) + static_cast(std::snprintf(buf.data(), buf.size(), "%.15g", d)); // NOLINT(cppcoreguidelines-pro-type-vararg,hicpp-vararg) break; case 2: - std::snprintf(buf.data(), buf.size(), "%.3e", d); // NOLINT(cppcoreguidelines-pro-type-vararg,hicpp-vararg) + static_cast(std::snprintf(buf.data(), buf.size(), "%.3e", d)); // NOLINT(cppcoreguidelines-pro-type-vararg,hicpp-vararg) break; case 3: - std::snprintf(buf.data(), buf.size(), "%.25g", d); // NOLINT(cppcoreguidelines-pro-type-vararg,hicpp-vararg) + static_cast(std::snprintf(buf.data(), buf.size(), "%.25g", d)); // NOLINT(cppcoreguidelines-pro-type-vararg,hicpp-vararg) break; default: - std::snprintf(buf.data(), buf.size(), "%.0f", d); // NOLINT(cppcoreguidelines-pro-type-vararg,hicpp-vararg) + static_cast(std::snprintf(buf.data(), buf.size(), "%.0f", d)); // NOLINT(cppcoreguidelines-pro-type-vararg,hicpp-vararg) break; } tokens.emplace_back(buf.data()); @@ -1232,7 +1234,7 @@ TEST_CASE("json_view values") const json j = json::parse(text); // user types with from_json, and other types, through basic_json - const record r = v.get(); + const auto r = v.get(); CHECK(r.name == "widget"); CHECK(r.count == 3); CHECK((v["pair"].get>() == j["pair"].get>())); @@ -1552,10 +1554,10 @@ TEST_CASE("json_view dump") { chars += "\xC3\xA9"; } - const json_document d = json_document::parse("{\"a\":\"" + chars + R"(","k\n":1})"); + const json_document d = json_document::parse(R"({"a":")" + chars + R"(","k\n":1})"); const json expected = json::parse("\"" + chars + "\""); CHECK(d.root()["a"].dump(-1, ' ', true) == expected.dump(-1, ' ', true)); - CHECK(d.root()["a"].dump(-1, ' ', true).size() == 2 + 5000 * 6); + CHECK(d.root()["a"].dump(-1, ' ', true).size() == 2 + (5000 * 6)); } SECTION("streams and discarded views") @@ -1647,7 +1649,7 @@ TEST_CASE("json_view comparison") // discarded values compare as basic_json's do const json discarded(json::value_t::discarded); CHECK((json_view() == json_view()) == (discarded == discarded)); // NOLINT(readability-container-size-empty): operator== is tested - CHECK((json_view() == discarded) == (discarded == discarded)); + CHECK((json_view() == discarded) == (discarded == discarded)); // NOLINT(readability-container-size-empty) const json_document null_document = json_document::parse("null"); CHECK(!(json_view() == null_document.root())); // NOLINT(readability-container-size-empty) CHECK(!(null_document.root() == discarded)); @@ -1736,6 +1738,7 @@ TEST_CASE("json_view large objects") ordered.push_back(keys[i]); } std::vector result; + result.reserve(ordered.size()); for (std::size_t i = 0; i < ordered.size(); ++i) { result.push_back({ordered[i], i}); @@ -1870,7 +1873,7 @@ TEST_CASE("json_view large objects") for (int object = 0; object < objects; ++object) { text += object != 0 ? ",{" : "{"; - for (int i = 0; i < members + object; ++i) + for (int i = 0, n = members + object; i < n; ++i) { text += (i != 0 ? ",\"" : "\"") + std::to_string(i) + "\":" + std::to_string(i); } @@ -1897,7 +1900,7 @@ TEST_CASE("json_view large objects") for (int object = 0; object < 5; ++object) { const json_view v = d.root()[static_cast(object)]; - for (int i = 0; i < 150 + object; ++i) + for (int i = 0, n = 150 + object; i < n; ++i) { CHECK(v[std::to_string(i)].get() == i); } diff --git a/tests/src/unit-json_view_builder.cpp b/tests/src/unit-json_view_builder.cpp index 8b50a1599..320d4dde9 100644 --- a/tests/src/unit-json_view_builder.cpp +++ b/tests/src/unit-json_view_builder.cpp @@ -302,7 +302,7 @@ TEST_CASE("json_view builder") { "1e39", "-1e39", "3.4028235e38", "-3.4028235e38", "3.4028236e38", "-3.4028236e38", "3.4028234663852886e38", "1e38", "340282356779733661637539395458142568448", "340282356779733661637539395458142568447.99", "0.00034028236e42", - "[1.5e38, 3.5e38]", "{\"a\": 1e-50, \"b\": 1e39}" + "[1.5e38, 3.5e38]", R"({"a": 1e-50, "b": 1e39})" }) { CAPTURE(text) @@ -419,7 +419,7 @@ TEST_CASE("json_view builder") }) { CAPTURE(name) - std::ifstream f(std::string(TEST_DATA_DIRECTORY) + name, std::ios::binary); + const std::ifstream f(std::string(TEST_DATA_DIRECTORY) + name, std::ios::binary); std::stringstream ss; ss << f.rdbuf(); const std::string text = ss.str(); diff --git a/tests/src/unit-msgpack.cpp b/tests/src/unit-msgpack.cpp index d29769738..aeb0116d3 100644 --- a/tests/src/unit-msgpack.cpp +++ b/tests/src/unit-msgpack.cpp @@ -2201,10 +2201,7 @@ struct huge_array : std::vector } }; -using huge_array_json = nlohmann::basic_json < - std::map, huge_array, std::string, bool, std::int64_t, std::uint64_t, - double, std::allocator, nlohmann::adl_serializer, - std::vector, void >; +using huge_array_json = nlohmann::json::with_array_t; TEST_CASE("MessagePack Size above uint32 for array") { @@ -2249,18 +2246,7 @@ template, - void >; +using huge_object_json = nlohmann::json::with_object_t; TEST_CASE("MessagePack Size above uint32 for object") { @@ -2295,18 +2281,7 @@ struct huge_string : std::string } }; -using huge_string_json = nlohmann::basic_json < - std::map, - std::vector, - huge_string, - bool, - std::int64_t, - std::uint64_t, - double, - std::allocator, - nlohmann::adl_serializer, - std::vector, - void >; +using huge_string_json = nlohmann::json::with_string_t; TEST_CASE("MessagePack Size above uint32 for string") { @@ -2329,18 +2304,7 @@ struct huge_binary : std::vector } }; -using huge_binary_json = nlohmann::basic_json < - std::map, - std::vector, - std::string, - bool, - std::int64_t, - std::uint64_t, - double, - std::allocator, - nlohmann::adl_serializer, - huge_binary, - void >; +using huge_binary_json = nlohmann::json::with_binary_t; TEST_CASE("MessagePack Size above uint32 for binary") { @@ -2390,14 +2354,10 @@ class beyond_uint32_string_t : public std::string } }; -using beyond_uint32_string_json = nlohmann::basic_json < - std::map, std::vector, beyond_uint32_string_t, bool, std::int64_t, std::uint64_t, - double, std::allocator, nlohmann::adl_serializer, std::vector, void >; +using beyond_uint32_string_json = nlohmann::json::with_string_t; #endif -using beyond_uint32_binary_json = nlohmann::basic_json < - std::map, std::vector, std::string, bool, std::int64_t, std::uint64_t, - double, std::allocator, nlohmann::adl_serializer, beyond_uint32_binary_t, void >; +using beyond_uint32_binary_json = nlohmann::json::with_binary_t; } // namespace TEST_CASE("MessagePack lengths beyond UINT32_MAX cannot be serialized") diff --git a/tests/src/unit-ordered_json2.cpp b/tests/src/unit-ordered_json2.cpp index 653820628..4f4e18844 100644 --- a/tests/src/unit-ordered_json2.cpp +++ b/tests/src/unit-ordered_json2.cpp @@ -217,16 +217,7 @@ void int_to_string(alt_string& target, std::size_t value) target = std::to_string(value).c_str(); } -using alt_json = nlohmann::basic_json < - std::map, - std::vector, - alt_string, - bool, - std::int64_t, - std::uint64_t, - double, - std::allocator, - nlohmann::adl_serializer >; +using alt_json = nlohmann::json::with_string_t; bool operator<(const char* op1, const alt_string& op2) noexcept {