From 6487678bc54986589ffca43954bc72821882ff5e Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Tue, 15 Sep 2026 20:46:28 +0200 Subject: [PATCH 01/64] Fix swap(array_t&)/swap(object_t&) to update parent pointers under JSON_DIAGNOSTICS (#5464) Both overloads swapped the underlying container storage but never called set_parents(), leaving elements moved into *this with stale m_parent pointers (typically nullptr from the free-standing array_t/object_t). This produced wrong JSON Pointer paths in diagnostic messages and could trip assert_invariant() on subsequent copies. Mirrors the fix already applied in swap(reference other). Signed-off-by: Niels Lohmann --- include/nlohmann/json.hpp | 2 ++ single_include/nlohmann/json.hpp | 2 ++ tests/src/unit-diagnostics.cpp | 31 +++++++++++++++++++++++++++++++ 3 files changed, 35 insertions(+) diff --git a/include/nlohmann/json.hpp b/include/nlohmann/json.hpp index 741c08b53..ca3cda17a 100644 --- a/include/nlohmann/json.hpp +++ b/include/nlohmann/json.hpp @@ -3607,6 +3607,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec { using std::swap; swap(*(m_data.m_value.array), other); + set_parents(); } else { @@ -3623,6 +3624,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec { using std::swap; swap(*(m_data.m_value.object), other); + set_parents(); } else { diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 2bb7a5b7b..d4972a35e 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -27376,6 +27376,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec { using std::swap; swap(*(m_data.m_value.array), other); + set_parents(); } else { @@ -27392,6 +27393,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec { using std::swap; swap(*(m_data.m_value.object), other); + set_parents(); } else { diff --git a/tests/src/unit-diagnostics.cpp b/tests/src/unit-diagnostics.cpp index 1e8ed6aa3..135ecf9d6 100644 --- a/tests/src/unit-diagnostics.cpp +++ b/tests/src/unit-diagnostics.cpp @@ -273,5 +273,36 @@ TEST_CASE("Regression tests for extended diagnostics") CHECK(j1["numbers"]["two"] == 2); CHECK(j1["string"] == "t"); } + + SECTION("Regression test - swap(array_t&)/swap(object_t&) must update JSON_DIAGNOSTICS parent pointers") + { + // swap(array_t&) + { + json j = json::array(); + json::array_t arr = {json::array({1})}; + j.swap(arr); + + // parent pointers of the moved-in elements must point into j, not + // into the now-defunct free-standing array_t + CHECK_THROWS_WITH_AS(j[0][0].get(), "[json.exception.type_error.302] (/0/0) type must be string, but is number", json::type_error); + + // must not trigger assert_invariant() in a debug/assert-enabled build + json const k = j; + CHECK(k == j); + } + + // swap(object_t&) + { + json o = json::object(); + json::object_t obj = {{"a", json::array({1})}}; + o.swap(obj); + + CHECK_THROWS_WITH_AS(o["a"][0].get(), "[json.exception.type_error.302] (/a/0) type must be string, but is number", json::type_error); + + // must not trigger assert_invariant() in a debug/assert-enabled build + json const p = o; + CHECK(p == o); + } + } } From c72f37a40d36220255ee9eb097e74ee8950a2681 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Tue, 15 Sep 2026 20:46:37 +0200 Subject: [PATCH 02/64] Support custom object/array types and improve template parameter handling (#5443) * docs: document the implicit requirements on basic_json's template parameters The requirements that basic_json places on its eleven template parameters were only implied by how the library uses the resulting object_t, array_t, string_t, etc. Consumers had to discover them by trial and error. Add "Template Parameter Requirements" collecting them, split into what is always required and what is only required when a particular part of the API is instantiated. Notable findings that were previously undocumented: - ObjectType must provide a key_compare member type (actual_object_comparator names object_t::key_compare in both arms of a std::conditional), and its third template parameter is used as a comparator, so std::unordered_map cannot be used without a wrapper. - ArrayType must provide capacity() -- push_back(), emplace_back(), operator+=(), and operator[](size_type) call it unconditionally -- and needs random-access iterators, so std::deque and std::list do not work. - StringType needs contiguous, null-terminated data(), a one-byte value_type, and either assignability from std::to_string or an ADL int_to_string(). - NumberFloatType must be float, double, or long double for parsing and serialization; the integer types must satisfy std::is_integral. - AllocatorType must be stateless, support incomplete types, and use plain pointers. - BooleanType and the number types are union members and must be trivial. Link the new page from the basic_json overview, the types feature page, and the individual type alias pages, and correct the container examples given for ObjectType (std::unordered_map) and ArrayType (std::list), which do not work. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_018hxZxz8svM54c6ATEvXp5E Signed-off-by: Niels Lohmann * Fix object_comparator_t for object types without key_compare detail::actual_object_comparator selected between object_t::key_compare and default_object_comparator_t with std::conditional. Both type arguments of std::conditional are named eagerly, so object_t::key_compare had to exist regardless of the condition, and the has_key_compare guard added in 3.11.0 never took effect: any ObjectType without a key_compare member type failed to compile while instantiating basic_json itself. Use detected_or_t instead, which resolves through a SFINAE partial specialization and only names object_t::key_compare when it exists. The selected type is unchanged for every object type that compiled before, so object_comparator_t -- a public member type -- keeps its meaning and ABI. has_key_compare had no other users and is removed. Add a regression test using an adapter around std::unordered_map, which has no key_compare; it fails to compile without this change. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_018hxZxz8svM54c6ATEvXp5E Signed-off-by: Niels Lohmann * docs: list the types that are known to work for each template parameter Follow up on the template parameter requirements page: state, for every template parameter, which concrete types work and where they stop working. Each entry was verified by compiling and running a common workload (DOM access, dump, parse, CBOR/MessagePack round-trip, flatten, hash) against that instantiation. Findings worth calling out: - ObjectType no longer needs a key_compare member type, so the std::unordered_map adapter only has to restore the template argument order. A hash-ordered ObjectType works everywhere except unflatten(), which reconstructs an array only when it meets the reference token 0 before the other indices. - ArrayType: std::deque works when wrapped to add capacity(); std::list does not. - StringType: std::pmr::string and std::basic_string with a custom allocator compile for the DOM, dump, and parse, but not for the binary readers, flatten, or diff, because the library assigns std::string values to string_t and int_to_string cannot be overloaded for a type in namespace std. - NumberFloatType: long double works for dump and parse but not for the binary formats, which have no encoding for it. - BinaryType: std::vector supports assignment, get, and the binary formats, but neither dump nor std::hash. Also record the object_comparator_t fix in its version history. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_018hxZxz8svM54c6ATEvXp5E Signed-off-by: Niels Lohmann * Fix unflatten and binary dumping for non-default configurations unflatten() decided between array and object by looking at the first reference token it happened to see for a node: it started an array only when that token was 0. With a sorted object type the token 0 always arrives first, so the result was correct by accident; with an object type whose iteration order is unspecified, {"/c/2":3,"/c/1":2,"/c/0":1} unflattened to an object with the keys "0", "1", and "2" instead of an array. Collect the pointer prefixes that have a reference token 0 among their children before building the result, and let get_and_create() consult that set. The outcome is now independent of the iteration order and matches, for every input, what a sorted object type produced before: a value is restored as an array if and only if one of its keys is 0. Iterating the flattened object in a different order would have been simpler, but it would have changed the key order of the result for insertion-ordered object types. The serializer, std::hash, and the UBJSON writer converted the elements of a binary value to an integer implicitly, which does not compile for a BinaryType whose value type is std::byte, and which made dump() write the bytes of a signed value type as negative numbers. Convert to std::uint8_t explicitly in all three places, so every byte type dumps as 0..255. The default std::vector configuration is unaffected. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_018hxZxz8svM54c6ATEvXp5E Signed-off-by: Niels Lohmann * docs: note which Abseil containers can be used as template arguments Checked against Abseil release 20250127.0 with the same workload as the other entries on the page (DOM access, dump, parse, CBOR/MessagePack/UBJSON round-trip, flatten, hash), with and without JSON_DIAGNOSTICS. absl::flat_hash_map and absl::node_hash_map work as ObjectType through an adapter that restores the template argument order and makes erase(iterator) return the following iterator, which Abseil's returns as void. The page now carries that adapter, and notes that absl::flat_hash_map does not keep references to the mapped values valid across insertions while absl::node_hash_map does. Both have a capacity() member, so JSON_DIAGNOSTICS already refreshes the parent pointers conservatively for them. absl::btree_map and absl::InlinedVector cannot be used at all: object_t and array_t are formed while basic_json is still incomplete, and both inspect their value type at class scope. std::map and std::vector are required by the standard to tolerate this, third-party containers generally are not, so the page states the constraint on its own rather than only per container. absl::InlinedVector does work as BinaryType, where it is instantiated with a complete type. absl::FixedArray and absl::Cord are not usable. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_018hxZxz8svM54c6ATEvXp5E Signed-off-by: Niels Lohmann * Relax the ArrayType and ObjectType requirements Two requirements forced users of otherwise suitable containers to write a wrapper, and neither was load-bearing. array_t::capacity() was read in push_back(), emplace_back(), operator+=(), and operator[](size_type), but set_parent() only looks at the value under JSON_DIAGNOSTICS; without diagnostics it was computed and discarded. Read it through array_capacity(), which reports unknown_size() when diagnostics are off or when the array type has no capacity() at all, and treat an unknown capacity as "the elements may have moved" so the parent pointers are refreshed conservatively. std::deque now works as ArrayType, in both builds, and capacity() is no longer named at all in a default build. Since the capacity is now only meaningful for array insertions, it moves out of set_parent() into set_parent_after_array_insert(). basic_json::erase(iterator) assigned the object's erase() return value, which requires the container to return the following iterator. Abseil's hash maps return void to avoid computing a successor the caller may not need. Detect that and compute the successor before erasing; containers that return an iterator, including the vector-backed ordered_map where a precomputed successor would be wrong, keep the existing path. Together these leave an Abseil hash map needing only an alias that restores the template argument order, and no adapter at all for std::deque. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_018hxZxz8svM54c6ATEvXp5E Signed-off-by: Niels Lohmann * Do not require string_t to be convertible from std::string Three places built a std::string and handed it to something expecting a string_t: the UBJSON high-precision number reader, which every binary reader instantiates, and the BSON writer's array element size calculation and write. That silently required string_t to be implicitly convertible from std::string, which std::string itself and types with a string_view conversion satisfy, but many string types do not. Construct the string_t explicitly from the data and size, which the requirements already cover. This makes boost::container::string, eastl::string, std::pmr::string, and std::basic_string with a custom allocator work as StringType, none of which could previously be used with any binary format. Add binary format coverage to the alt_string test, which had none, including a UBJSON high-precision number -- the case that goes through the reader path. BSON stays uncovered there: it additionally needs string_t::find(value_type), which alt_string does not provide. Also record which containers from Boost, Abseil, and EASTL work for each template parameter, and correct two claims: std::pmr::string is usable after this change, and tsl::ordered_map is not usable at all, because its iterators expose the mapped value as const while basic_json modifies it in place. Signed-off-by: Niels Lohmann * docs: record compatibility for the common header-only hash maps ankerl::unordered_dense (map and segmented_map), phmap (flat_hash_map and node_hash_map), and robin_hood::unordered_flat_map all work as ObjectType through the same adapter as Abseil's and Boost's hash maps, which only has to restore the template argument order. phmap::btree_map and robin_hood::unordered_node_map do not: like the other btree containers they require a complete value type. Note that none of these hash maps defines key_compare, so every one of them depends on object_comparator_t falling back to default_object_comparator_t. Signed-off-by: Niels Lohmann * docs: record Folly and the remaining vector replacements Folly works, with the caveat that its headers need C++20: folly::fbstring as StringType, folly::fbvector and folly::small_vector as ArrayType, folly::fbvector as BinaryType, and folly::F14NodeMap as ObjectType through the usual argument-order adapter. folly::F14FastMap is the exception and requires a complete value type. For ArrayType, boost::container::devector, boost::container::static_vector (within its fixed capacity), and std::pmr::vector work as well. Signed-off-by: Niels Lohmann * docs: cover fifo_map, gtl, folly::sorted_vector_map, and Qt nlohmann::fifo_map works through the adapter that has always been documented for it, and preserves the insertion order. Restore its mention in the object order page, which was dropped together with the tsl::ordered_map one: unlike ordered_map it keeps a lookup index, so it is the insertion-ordered option without the quadratic cost. gtl::flat_hash_map and folly::sorted_vector_map work as well, the latter through an alias that drops the allocator, whose value type it disagrees on. gtl::btree_map does not, for the same reason as the other btree containers. None of the Qt containers can be used, each for its own reason: QMap has no value_type, QHash iterators yield the mapped value rather than a pair, QList has no max_size(), QByteArray spells empty() as isEmpty(), and QString is UTF-16. Signed-off-by: Niels Lohmann * docs: qualify the std::pmr::string support claim Listing std::pmr::string as fully supported was an overclaim: it was only ever checked with the default memory resource, which is not what PMR is for. basic_json cannot be given an allocator or a memory resource, so a pmr string inside a value always allocates from std::pmr::get_default_resource(), and assigning an arena-backed string into a value silently drops its resource, because polymorphic_allocator does not propagate on copy construction. Passing polymorphic_allocator as AllocatorType does not compile either. Only the process-global set_default_resource() redirects these allocations. Say so, and separate the row from std::basic_string with a custom stateless allocator, which is unaffected. Signed-off-by: Niels Lohmann * docs: remove a duplicated StringType compatibility section The StringType section carried two 'Compatible types' tables and two copies of the reference-implementation tip. The second table was a stale copy from before the binary format string fixes and still listed std::pmr::string and std::basic_string with a custom allocator as unusable, contradicting the corrected table a few lines above it, and it dragged along the old explanation that blamed int_to_string. Drop the stale copy and put the surviving table before the notes, so the 'see below' in the std::pmr::string row points forwards. Signed-off-by: Niels Lohmann * docs: correct the template parameter requirements after independent verification Every claim on the page was re-checked by compiling and running it, including the rows that say a type cannot be used, which were checked to fail for the documented reason and not merely to fail. Twenty-four claims were wrong. The most consequential: the incomplete-type constraint applies to ObjectType only. object_t is instantiated inside the class definition, because it is probed for key_compare; array_t is only named there and is not instantiated until basic_json is complete. So eastl::vector, QList and QVector are not excluded by incomplete types at all -- they simply have no max_size() -- and absl::InlinedVector is excluded for a subtler reason of its own. Further corrections: ObjectType does not need erase(key), which has a fallback, but does need at(key) for UBJSON output; only == and < are used, or == and <=> under C++20, not all six; the documented adapter does not fit ankerl or robin_hood. ArrayType needs no initializer-list insert, and value_type, the (count, value) constructor and swappability are per-function, not always. BinaryType needs a range insert for CBOR indefinite-length byte strings and does not need push_back. StringType needs append(const StringType&) unconditionally, and does not need operator!= or operator== against const char*; empty(), resize(n) and reserve(n) are per-subsystem; int_to_string is needed by diff, items and std::hash rather than by JSON Pointer or flatten. BooleanType must be implicitly convertible from bool, and JSONSerializer's second parameter need not carry a default. std::pmr::string was wrong in the other direction this time: a moved-in string does keep its memory resource, and later growth allocates from it. Only copies land on the default resource. Five requirement violations are not caught at compile time rather than the two the page claimed; they are now listed together up front. Split every compatibility table into what works and what does not, as the reasons in the second half are the useful part. Signed-off-by: Niels Lohmann * Reduce the string_t and array_t members the library requires Several members were required only because of how the library happened to be written, not because the functionality needs them. Dropping them widens the set of usable string and array types, and one of them was also a performance problem. string_t: - c_str() is gone. Every call site already knew the length and passed it along, so data() is enough. The one place that did not, the diagnostics path in exceptions.hpp, now builds the token from data() and size(), which also stops it from truncating keys that contain a null byte. - back() is gone; the serializer indexes the last character instead. - find(str, pos), replace(), and substr() are gone. escape() and unescape() rebuilt the string with one replace() per escaped character, which moves the tail every time: escaping a string of n characters that all need escaping cost O(n^2). Both now scan with find_first_of() -- a member the pointer parser already required -- and append whole runs, so the common case is one search and one copy. Escaping 64000 tildes drops from 717 ms to 20 ms; a string with nothing to escape gets faster too (8.4 ms to 5.8 ms), because the scan is still a single memchr per pass. json_pointer::split() takes its reference tokens with the (const char*, size_type) constructor rather than substr(). - json_pointer::to_string() accumulates with concat instead of letting concat default to std::string and converting afterwards, so streaming a json_pointer no longer requires string_t to be assignable from a std::string. array_t: - at(size_type) is gone. basic_json::at(size_type) checked the index by calling array_t::at() and translating std::out_of_range, which also required the array type to throw that exact exception. It now compares against size() and uses operator[]. The thrown exception, its message, and the behaviour under JSON_NOEXCEPTION are unchanged. The BSON writer wrote the terminating null byte out of the string's own buffer (size() + 1). It now writes the byte itself, so string_t::data() need not be null-terminated for to_bson(). The tests pin the reduced API: alt_string loses the five dropped members and gains coverage of the escaping paths, and a std::vector whose at() is hidden is used as an ArrayType. Signed-off-by: Niels Lohmann * docs: record the reduced string_t and array_t requirements Drop c_str(), back(), find(str, pos), replace(), and substr() from the StringType requirements and at(size_type) from the ArrayType ones, and note the string assignment the JSON pointer code performs. Streaming a json_pointer no longer needs assignability from a std::string. Add the non-null-terminated data() to the list of violations that are not diagnosed at compile time -- it was described in the StringType section but missing from the summary at the top -- and correct the QString row, which no longer fails for the c_str() it lacks. JSON_CATCH_USER no longer wraps a catch of std::out_of_range: the last one went away with array_t::at(). Describe what the library actually catches. Signed-off-by: Niels Lohmann * Use character literals for the signed BinaryType test MSVC rejects char(0xFF) with C4310 (cast truncates constant value), which the Windows workflow treats as an error. The character literals carry the same byte values without a narrowing cast. Signed-off-by: Niels Lohmann * Do not instantiate a hash map with an incomplete basic_json in the tests object_t is probed for key_compare inside the definition of basic_json, so it is instantiated while basic_json is still incomplete. Whether a hash map survives that depends on the standard library: libstdc++ 9 needs the size of the mapped type to instantiate std::unordered_map's node type and rejects the adapter, which broke the GCC 9 builds. The test now derives its no-key_compare object type from std::map -- which does cope -- and shadows the inherited key_compare member type with an entity that is not a type, so the library's probe finds none, exactly as for a hash map. The unflatten() order-independence checks in unit-json_pointer already cover the behaviour that the unordered object type was there for. The limitation is documented for std::unordered_map. Also address two Clang-Tidy findings the earlier commits introduced: erase_from_object() declares its iterator with auto, and at(size_type) checks the type first and then falls through to the return instead of throwing from an else branch. Signed-off-by: Niels Lohmann * Keep diagnostic key paths null-terminated Building the token from data() and size() kept an embedded null byte in the key, and since what() hands out a C string, that truncated the whole message rather than just the key: to_bson() on a key containing U+0000 reported "[json.exception.out_of_range.409] (/en" instead of the full explanation. This broke test-bson under JSON_DIAGNOSTICS. Constructing from data() alone stops at the first null byte, which is what c_str() did before, so the message is unchanged -- without requiring string_t to provide c_str(). Signed-off-by: Niels Lohmann * Do not parse the value in the array-at() test JSON_DIAGNOSTIC_POSITIONS adds the byte range of the value to the exception message, which a parsed value has and an in-memory one does not, so the two message checks failed in that configuration. Build the array in memory instead of parsing it; the test is about at(size_type) not needing array_t::at(), and the byte range is beside the point. Signed-off-by: Niels Lohmann * Move the custom BinaryType tests into their own translation unit The two sections added to unit-regression2.cpp brought a third full basic_json instantiation into a translation unit that was already large. With Clang on MinGW that pushed the object over the reach of a 32-bit relocation and test-regression2_cpp20.exe failed to link: relocation truncated to fit: IMAGE_REL_AMD64_REL32 against `.rdata' unit-regression2.cpp is restored to exactly what it was before, and the coverage moves to unit-custom-binary-type.cpp, next to the object and array type tests it belongs with. The signed value type is now also covered in C++11, where std::byte is not available. Signed-off-by: Niels Lohmann * Do not require the container iterators to be nothrow move constructible iter_impl declared its defaulted move operations noexcept. The exception specification a defaulted function gets implicitly follows from its members, here internal_iterator, which holds the object and array iterators. libstdc++ gives std::deque's iterator a user-provided copy constructor without noexcept before version 11, so the implicit specification is noexcept(false) and does not match the declared one. That deletes the function -- and with g++ 4.8, which predates CWG 1778, it is an error outright: error: function 'iter_impl >::iter_impl( iter_impl&&)' defaulted on its first declaration with an exception-specification that differs from the implicit declaration So std::deque, which this branch documents as a usable array type, could not be used with an older standard library. Leaving the specification to be computed cannot mismatch; iteration_proxy_value already spells out the same condition next door. The default configuration is unaffected: json::iterator, json::const_iterator and ordered_json::iterator stay nothrow move constructible and move assignable, which the test now checks so it cannot regress unnoticed. Signed-off-by: Niels Lohmann * Address two Clang-Tidy findings the custom container tests exposed Both come from instantiating basic_json with containers other than the default ones, and neither shows up with the Clang-Tidy version available outside CI: - insert(const_iterator, basic_json&&) forwards its by-value iterator to the const-reference overload. performance-unnecessary-value-param asks for the copy to be a move; it only fires for an iterator that is not trivially copyable, as std::deque's is not. The NOLINT on the function does not cover it, because the finding is reported where the parameter is used rather than where it is declared. Move it, which is what the check asks for and is a (very small) improvement in its own right. - cppcoreguidelines-use-enum-class rejects the unnamed enum that shadowed the inherited key_compare member type. An enum class would not do, since it declares a type of that name and the probe would find it again; a member function declaration hides the name just as well. Signed-off-by: Niels Lohmann * Assert the iterators' exception specification relative to the container The test pinned that nlohmann::json's iterators stay nothrow movable after iter_impl's defaulted move operations lost their declared noexcept. That is not a property of the library, though: the exception specification is now computed from the container iterators, so it holds only for standard library implementations whose iterators are themselves nothrow movable. MSVC's checked iterators before VS2017 are not -- _Iterator_base12 registers the iterator with the container's debug proxy in a copy constructor that carries no noexcept -- so the assertions fail on a Visual Studio 2015 debug build, which is the one debug configuration in the AppVeyor matrix and has no counterpart in the GitHub Actions matrix. Assert what the change actually guarantees instead: the iterators are nothrow movable exactly when the object and array iterators they are built from are. That still pins the default configuration against a silent regression, and it is true whatever the standard library provides. Signed-off-by: Niels Lohmann * Detect a void-returning erase() through a named trait erase_from_object() distinguished its two overloads with a decltype of a member call written inline in a default template argument. Every other detection in the library goes through the detector machinery in detected.hpp instead -- has_erase_with_key_type is the same question about the same member function -- and the inline form is the one shape older compilers are least reliable about. Express it the same way: detect_erase_with_iterator plus is_detected_exact, both of which the library already relies on elsewhere. No behaviour changes. Signed-off-by: Niels Lohmann * Give the custom container types only the constructors the library uses The three container types in the new tests inherited every constructor of their base with using Base::Base. That asks for more than the test needs: the library builds an object or an array by default construction, by copy or move, and -- when converting between two basic_json types or from an initializer list -- from an iterator range. Declaring those directly makes the requirement visible in the test, and keeps object types out of a corner where a compiler has to declare std::map's whole constructor set for a derived class while basic_json is still incomplete. Signed-off-by: Niels Lohmann * Temporarily disable the new custom container tests AppVeyor is the only CI that builds MSVC 2015 and 2017, and it has now rejected three heads of this branch. Its build log is not reachable from where this is being worked on, so the verdict is a single bit and the cause has to be narrowed down by bisection. Everything else stays: the library changes, the reduced alt_string, and the unflatten() tests. If AppVeyor passes with these three translation units disabled, the cause is one of the six basic_json instantiations they add; if it fails, it is in the library. Either way this commit is reverted. Signed-off-by: Niels Lohmann * Guard the disabled tests with a macro rather than #if 0 Clang-Tidy's readability-avoid-unconditional-preprocessor-if rejects a literal #if 0. Use a macro that is never defined instead, which the check does not look at. Still temporary, and reverted together with the previous commit. Signed-off-by: Niels Lohmann * Re-enable the array and binary container tests AppVeyor passed with all three new translation units disabled, so the library changes, the reduced alt_string, and the unflatten() tests are fine on MSVC 2015 and 2017; the cause is one of the six basic_json instantiations the new tests add. Bring back two of the three. If AppVeyor passes again, the cause is in unit-custom-object-type.cpp, which is the one still disabled; if it fails, it is in one of these two and needs one more split. Still temporary. Signed-off-by: Niels Lohmann * Diagnose two silently violated template parameter requirements Both were on the list of requirements that are not caught at compile time and corrupt values rather than failing, and both are a plain size comparison: - A BinaryType whose value_type is wider than one byte, which the readers and writers reinterpret as raw bytes anyway. - A NumberUnsignedType too narrow to hold the absolute value of every NumberIntegerType value, which makes basic_json(INT64_MIN).dump() yield -0 for std::int64_t with std::uint32_t. Neither static_assert rejects a configuration that worked before: both only fire where the result was already wrong. Also add the two comments the review asked for, in write_bson_string() and calc_bson_array_size(), matching the ones their counterparts already carry. Signed-off-by: Niels Lohmann * Align the template parameter tables and record what is now diagnosed Every table in the page is reformatted so each column is exactly as wide as its widest cell, which is what the review asked for in a dozen places: the separator rows that ran two dashes long, the stray spaces, and the columns padded well past their content. The row listing six containers that require a complete mapped type is split in two so that one cell no longer sets the width of the whole table. Content changes: NumberUnsignedType is described as any unsigned integer type at least as wide as NumberIntegerType rather than any unsigned integer type; the two requirements that are now static_asserts move out of the list of violations that are not caught at compile time; and the two places that require a non-const operator[] say why data() will not do (std::string has no non-const data() before C++17). Signed-off-by: Niels Lohmann * Bisect the other way: only the object container tests The previous head touched only docs/, which AppVeyor's only_commits filter skips, so it produced no build and no status at all -- the pull request looked green without ever having been built on MSVC 2015 or 2017. Swap the guards instead of repeating that step: unit-custom-object-type.cpp is enabled and the array and binary translation units are disabled. AppVeyor already passed with all three disabled, so a failure here pins the cause on no_key_compare_json or void_erase_json, and a pass pins it on the array or binary file. Still temporary. Signed-off-by: Niels Lohmann * Split the two object types apart AppVeyor failed with only unit-custom-object-type.cpp enabled and passed with all three new translation units disabled, so the cause is one of the two object types in this file and not the array or binary ones. Guard out void_erase_map and leave no_key_compare_map, which separates the two constructs under suspicion: shadowing the inherited key_compare member type with an entity that is not a type, and hiding the inherited erase with a void-returning overload. A failure here points at the first, a pass at the second. Still temporary. Signed-off-by: Niels Lohmann * Build the no-key_compare object type by composition, not inheritance The "object type without key_compare" test failed on AppVeyor's MSVC 2017 jobs (/std:c++17): its no_key_compare_map derived publicly from std::map and shadowed the inherited key_compare type with a same-named member function, relying on ordinary member hiding to make key_compare unreachable as a type for the library's detection trait. MSVC 2017 does not honor that hiding for a typename-qualified lookup performed from outside the class and still resolves key_compare to the base's comparator type, so object_comparator_t incorrectly picked it up instead of falling back to default_object_comparator_t. Wrapping a std::map by composition instead removes the base class entirely, so there is no key_compare to find under any lookup rule, on any compiler. Also drops the now-unneeded JSON_BISECT_CUSTOM_CONTAINER_TESTS guard left over from narrowing this down: the void_erase_map test in the same file was never the cause and is re-enabled unconditionally. Verified locally with clang++ and g++ under C++17 and C++20. Signed-off-by: Niels Lohmann * Re-enable the array and binary custom-container tests unit-custom-array-type.cpp and unit-custom-binary-type.cpp were still guarded behind JSON_BISECT_CUSTOM_CONTAINER_TESTS from bisecting the AppVeyor failure fixed in 7c39f3227, which was unrelated to either file. The macro was never defined, so none of these tests actually ran in CI. Verified locally with clang++ and g++ under C++17 and C++20 before removing the guards. Signed-off-by: Niels Lohmann * Fix indentation of custom_object_type per astyle The one-line function bodies in the composition-based no_key_compare_map (7c39f3227) do not match the project's Allman brace style, which the ci_test_amalgamation job enforces with astyle. Reformatted with the pinned astyle 3.4.13; no functional change. Signed-off-by: Niels Lohmann * docs: add compiled reference implementations for the container template parameters Each of ObjectType, ArrayType, StringType, and BinaryType now links to a minimal, self-contained header (docs/mkdocs/docs/examples/custom_*_type.hpp) that wraps the corresponding standard container by composition and satisfies every "Always required" member listed on that page. Unlike the prose requirement lists, these are real code: each header has a companion .cpp that instantiates a basic_json specialization with it and is compiled and run by the existing ci_test_examples check (docs/Makefile's check_output_portable), so the reference implementations cannot silently drift from what the library actually requires. The .output files were generated with that same target. StringType's existing pointer to tests/src/unit-alt-string.cpp's alt_string is kept alongside the new header as a more thorough, battle-tested example. Verified locally: astyle (pinned 3.4.13, project .astylerc) on the new files; clang++/g++ under C++11/17/20 for each example against the amalgamated header; `make check_output_portable` in docs/; `mkdocs build --strict` and scripts/check_structure.py for the page itself. Signed-off-by: Niels Lohmann * Declare no_key_compare_map's accessors noexcept The GCC C++20 job builds with -Wnoexcept and -Werror, and the standard library takes noexcept(c.begin()) and noexcept(c.end()) in ranges_base.h and range_access.h. Forwarding to std::map without repeating its noexcept made those expressions false, which the warning reports as an error: error: noexcept-expression evaluates to 'false' because of a call to no_key_compare_map<...>::begin() [-Werror=noexcept] note: but ... does not throw; perhaps it should be declared 'noexcept' Give the accessors the exception specification of what they forward to. std::map declares begin, end, cbegin, cend, empty, size, max_size, and clear noexcept, so the wrapper does too. swap is left alone: std::map's is only conditionally noexcept, and nothing asks for it. void_erase_map is unaffected because it still derives from std::map and inherits accessors that already carry the specification. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_018hxZxz8svM54c6ATEvXp5E Signed-off-by: Niels Lohmann * Write out what the defaulted constructor of no_key_compare_map implied ci_test_gcc builds with -Weffc++, which asks for data to be initialized in a member initialization list; a defaulted default constructor does not do that: error: 'no_key_compare_map<...>::data' should be initialized in the member initialization list [-Werror=effc++] Writing the constructor out satisfies that but drops the exception specification the defaulted one carried, which -Wnoexcept then objects to where the standard library takes noexcept(construct(...)). Declare it the way the defaulted constructor was: noexcept when the wrapped map's default constructor is. This is the cost of composition -- inheritance carried std::map's exception specifications and initialization for free, and forwarding by hand has to restate them. Checked with the repository's own GCC warning set from cmake/gcc_flags.cmake, all 346 flags, at C++11, C++17 and C++20: no diagnostics for this file, nor for the two custom container translation units that were disabled while the MSVC failure was narrowed down and are built again now. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_018hxZxz8svM54c6ATEvXp5E Signed-off-by: Niels Lohmann * Mark no_key_compare_map::swap noexcept Clang-Tidy rejects a swap that is not: error: swap functions should be marked noexcept [cppcoreguidelines-noexcept-swap,performance-noexcept-swap] It was left unmarked on the grounds that std::map::swap is only conditionally noexcept, so an unconditional promise would be wrong for a comparator or allocator that can throw while swapping. Both concerns are met by taking the specification from the wrapped map rather than asserting one: noexcept(noexcept(data.swap(other.data))). Clang-Tidy accepts that, and no NOLINT is needed. Last in the series of specifications that inheritance used to supply and composition has to write out by hand. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_018hxZxz8svM54c6ATEvXp5E Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann Co-authored-by: Claude Opus 5 --- docs/mkdocs/docs/api/basic_json/array_t.md | 7 +- docs/mkdocs/docs/api/basic_json/binary_t.md | 12 +- docs/mkdocs/docs/api/basic_json/boolean_t.md | 8 + docs/mkdocs/docs/api/basic_json/index.md | 4 + .../docs/api/basic_json/json_base_class_t.md | 5 +- .../docs/api/basic_json/json_serializer.md | 6 + .../docs/api/basic_json/number_float_t.md | 10 + .../docs/api/basic_json/number_integer_t.md | 7 + .../docs/api/basic_json/number_unsigned_t.md | 8 + .../api/basic_json/object_comparator_t.md | 2 + docs/mkdocs/docs/api/basic_json/object_t.md | 7 +- docs/mkdocs/docs/api/basic_json/string_t.md | 7 + docs/mkdocs/docs/api/basic_json/unflatten.md | 10 +- .../mkdocs/docs/api/macros/json_throw_user.md | 8 +- .../docs/examples/custom_array_type.cpp | 19 + .../docs/examples/custom_array_type.hpp | 152 ++++ .../docs/examples/custom_array_type.output | 2 + .../docs/examples/custom_binary_type.cpp | 21 + .../docs/examples/custom_binary_type.hpp | 112 +++ .../docs/examples/custom_binary_type.output | 2 + .../docs/examples/custom_object_type.cpp | 26 + .../docs/examples/custom_object_type.hpp | 144 ++++ .../docs/examples/custom_object_type.output | 11 + .../docs/examples/custom_string_type.cpp | 20 + .../docs/examples/custom_string_type.hpp | 134 ++++ .../docs/examples/custom_string_type.output | 10 + docs/mkdocs/docs/features/object_order.md | 6 +- docs/mkdocs/docs/features/types/index.md | 7 +- .../features/types/template_parameters.md | 747 ++++++++++++++++++ docs/mkdocs/mkdocs.yml | 1 + include/nlohmann/detail/exceptions.hpp | 5 +- include/nlohmann/detail/hash.hpp | 4 +- .../nlohmann/detail/input/binary_reader.hpp | 5 +- .../nlohmann/detail/iterators/iter_impl.hpp | 9 +- include/nlohmann/detail/json_pointer.hpp | 55 +- include/nlohmann/detail/meta/type_traits.hpp | 29 +- .../nlohmann/detail/output/binary_writer.hpp | 38 +- include/nlohmann/detail/output/serializer.hpp | 26 +- include/nlohmann/detail/string_escape.hpp | 99 ++- include/nlohmann/json.hpp | 151 ++-- single_include/nlohmann/json.hpp | 421 +++++++--- tests/src/unit-alt-string.cpp | 72 +- tests/src/unit-custom-array-type.cpp | 150 ++++ tests/src/unit-custom-binary-type.cpp | 79 ++ tests/src/unit-custom-object-type.cpp | 323 ++++++++ tests/src/unit-json_pointer.cpp | 10 + 46 files changed, 2721 insertions(+), 270 deletions(-) create mode 100644 docs/mkdocs/docs/examples/custom_array_type.cpp create mode 100644 docs/mkdocs/docs/examples/custom_array_type.hpp create mode 100644 docs/mkdocs/docs/examples/custom_array_type.output create mode 100644 docs/mkdocs/docs/examples/custom_binary_type.cpp create mode 100644 docs/mkdocs/docs/examples/custom_binary_type.hpp create mode 100644 docs/mkdocs/docs/examples/custom_binary_type.output create mode 100644 docs/mkdocs/docs/examples/custom_object_type.cpp create mode 100644 docs/mkdocs/docs/examples/custom_object_type.hpp create mode 100644 docs/mkdocs/docs/examples/custom_object_type.output create mode 100644 docs/mkdocs/docs/examples/custom_string_type.cpp create mode 100644 docs/mkdocs/docs/examples/custom_string_type.hpp create mode 100644 docs/mkdocs/docs/examples/custom_string_type.output create mode 100644 docs/mkdocs/docs/features/types/template_parameters.md create mode 100644 tests/src/unit-custom-array-type.cpp create mode 100644 tests/src/unit-custom-binary-type.cpp create mode 100644 tests/src/unit-custom-object-type.cpp diff --git a/docs/mkdocs/docs/api/basic_json/array_t.md b/docs/mkdocs/docs/api/basic_json/array_t.md index dd2b901d5..b6e41a6ad 100644 --- a/docs/mkdocs/docs/api/basic_json/array_t.md +++ b/docs/mkdocs/docs/api/basic_json/array_t.md @@ -14,7 +14,11 @@ To store objects in C++, a type is defined by the template parameters explained ## Template parameters `ArrayType` -: container type to store arrays (e.g., `std::vector` or `std::list`) +: container type to store arrays. It must be a vector-like container: the library uses `operator[]`, `at()`, and + `resize()`, and requires random-access iterators. `#!cpp std::vector` and `#!cpp std::deque` qualify; + `#!cpp std::list` does not. See + [Template Parameter Requirements](../../features/types/template_parameters.md#arraytype) for the full list of + requirements. `AllocatorType` : the allocator to use for objects (e.g., `std::allocator`) @@ -66,3 +70,4 @@ Arrays are stored as pointers in a `basic_json` type. That is, for any access to ## Version history - Added in version 1.0.0. +- Made `capacity()` optional, so that array types such as `#!cpp std::deque` can be used, in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json/binary_t.md b/docs/mkdocs/docs/api/basic_json/binary_t.md index 64506b25f..36600748a 100644 --- a/docs/mkdocs/docs/api/basic_json/binary_t.md +++ b/docs/mkdocs/docs/api/basic_json/binary_t.md @@ -42,7 +42,9 @@ represent a byte array in modern C++. `value_type` must additionally be exactly one byte wide (e.g., `std::uint8_t`/`char`/`std::byte`): the binary serializers (CBOR, MessagePack, BSON, UBJSON) read and write the container's raw bytes via `reinterpret_cast`, which is only correct for byte-sized elements -- a container like - `#!cpp std::vector` will not work as `BinaryType`. + `#!cpp std::vector` will not work as `BinaryType`. The elements must be stored contiguously, and + the binary readers additionally require `resize()` and `operator[]`. See + [Template Parameter Requirements](../../features/types/template_parameters.md#binarytype) for the full list. ## Notes @@ -50,6 +52,11 @@ represent a byte array in modern C++. The default values for `BinaryType` is `#!cpp std::vector`. +#### Supported byte types + +`#!cpp std::vector`, `#!cpp std::vector`, and `#!cpp std::vector` are supported. +Regardless of which of them is configured, [`dump`](dump.md) writes the bytes as the numbers 0..255. + #### Custom BinaryType behavior When a custom `BinaryType` is configured (other than the default `#!cpp std::vector`), you can assign @@ -126,3 +133,6 @@ type `#!cpp binary_t*` must be dereferenced. ## Version history - Added in version 3.8.0. Changed the type of subtype to `std::uint64_t` in version 3.10.0. +- Fixed [`dump`](dump.md), [`std::hash`](std_hash.md), and [`to_ubjson`](to_ubjson.md) for byte types that are not + integers (e.g., `#!cpp std::byte`) in version 3.13.0. `dump` now writes the bytes of a signed byte type (e.g., + `#!cpp char`) as 0..255 rather than as negative numbers. diff --git a/docs/mkdocs/docs/api/basic_json/boolean_t.md b/docs/mkdocs/docs/api/basic_json/boolean_t.md index c30afefdc..bfb8c3426 100644 --- a/docs/mkdocs/docs/api/basic_json/boolean_t.md +++ b/docs/mkdocs/docs/api/basic_json/boolean_t.md @@ -11,6 +11,14 @@ literals `#!json true` and `#!json false`. To store boolean values in C++, a type is defined by the template parameter `BooleanType` which chooses the type to use. +## Template parameters + +`BooleanType` +: the type to store booleans. As it is stored directly inside a `basic_json` value (in a union), it must be a + trivially default-constructible, trivially copyable, and trivially destructible type that is convertible to and + from `#!cpp bool`. See + [Template Parameter Requirements](../../features/types/template_parameters.md#booleantype). + ## Notes #### Default type diff --git a/docs/mkdocs/docs/api/basic_json/index.md b/docs/mkdocs/docs/api/basic_json/index.md index bd33f31ac..c5556ea3b 100644 --- a/docs/mkdocs/docs/api/basic_json/index.md +++ b/docs/mkdocs/docs/api/basic_json/index.md @@ -35,6 +35,10 @@ class basic_json; | `BinaryType` | type for binary arrays | [`binary_t`](binary_t.md) | | `CustomBaseClass` | extension point for user code | [`json_base_class_t`](json_base_class_t.md) | +The library imposes a number of requirements on these types that are not expressed as C++ concepts, such as the +container operations `object_t` and `array_t` must provide, or the fact that `StringType` must be `char`-based. They +are collected in [Template Parameter Requirements](../../features/types/template_parameters.md). + ## Specializations - [**json**](../json.md) - default specialization diff --git a/docs/mkdocs/docs/api/basic_json/json_base_class_t.md b/docs/mkdocs/docs/api/basic_json/json_base_class_t.md index dfe5f1cb7..7add54098 100644 --- a/docs/mkdocs/docs/api/basic_json/json_base_class_t.md +++ b/docs/mkdocs/docs/api/basic_json/json_base_class_t.md @@ -21,8 +21,11 @@ The default value for `CustomBaseClass` is `void`. In this case, an #### Limitations -The type `CustomBaseClass` has to be a default-constructible class. +The type `CustomBaseClass` has to be a default-constructible, non-`final` class. `basic_json` only supports copy/move construction/assignment if `CustomBaseClass` does so as well. +A `CustomBaseClass` with non-static data members forfeits `basic_json`'s +[standard layout](https://en.cppreference.com/w/cpp/named_req/StandardLayoutType) guarantee. See +[Template Parameter Requirements](../../features/types/template_parameters.md#custombaseclass). ## Examples diff --git a/docs/mkdocs/docs/api/basic_json/json_serializer.md b/docs/mkdocs/docs/api/basic_json/json_serializer.md index 24a37735c..b92cf3d17 100644 --- a/docs/mkdocs/docs/api/basic_json/json_serializer.md +++ b/docs/mkdocs/docs/api/basic_json/json_serializer.md @@ -19,6 +19,12 @@ using json_serializer = JSONSerializer; The default values for `json_serializer` is [`adl_serializer`](../adl_serializer/index.md). +#### Requirements + +A custom serializer must provide `#!cpp static void to_json(basic_json&, T)` for every type it serializes, and either +`#!cpp static void from_json(const basic_json&, T&)` or `#!cpp static T from_json(const basic_json&)` for every type it +deserializes. See [Template Parameter Requirements](../../features/types/template_parameters.md#jsonserializer). + ## Examples ??? example diff --git a/docs/mkdocs/docs/api/basic_json/number_float_t.md b/docs/mkdocs/docs/api/basic_json/number_float_t.md index 3e8933da6..83c7011c5 100644 --- a/docs/mkdocs/docs/api/basic_json/number_float_t.md +++ b/docs/mkdocs/docs/api/basic_json/number_float_t.md @@ -20,6 +20,16 @@ used. To store floating-point numbers in C++, a type is defined by the template parameter `NumberFloatType` which chooses the type to use. +## Template parameters + +`NumberFloatType` +: the type to store floating-point numbers. Parsing and serialization are implemented in terms of + `#!cpp std::strtof`/`#!cpp std::strtod`/`#!cpp std::strtold` and `#!cpp std::snprintf`, so the type must be + `#!cpp float`, `#!cpp double`, or `#!cpp long double`. The + [binary formats](../../features/binary_formats/index.md) additionally require `#!cpp float` or `#!cpp double`, + because they have no encoding for `#!cpp long double`. See + [Template Parameter Requirements](../../features/types/template_parameters.md#numberfloattype). + ## Notes #### Default type diff --git a/docs/mkdocs/docs/api/basic_json/number_integer_t.md b/docs/mkdocs/docs/api/basic_json/number_integer_t.md index 79cbdf8ca..9a2ffab7f 100644 --- a/docs/mkdocs/docs/api/basic_json/number_integer_t.md +++ b/docs/mkdocs/docs/api/basic_json/number_integer_t.md @@ -20,6 +20,13 @@ used. To store integer numbers in C++, a type is defined by the template parameter `NumberIntegerType` which chooses the type to use. +## Template parameters + +`NumberIntegerType` +: the type to store signed integers. It must be a **signed integral** type (`#!cpp std::is_integral`) with a + `#!cpp std::numeric_limits` specialization, and it is stored directly inside a `basic_json` value. See + [Template Parameter Requirements](../../features/types/template_parameters.md#numberintegertype-and-numberunsignedtype). + ## Notes #### Default type diff --git a/docs/mkdocs/docs/api/basic_json/number_unsigned_t.md b/docs/mkdocs/docs/api/basic_json/number_unsigned_t.md index f1010f2a6..674f7711d 100644 --- a/docs/mkdocs/docs/api/basic_json/number_unsigned_t.md +++ b/docs/mkdocs/docs/api/basic_json/number_unsigned_t.md @@ -20,6 +20,14 @@ used. To store unsigned integer numbers in C++, a type is defined by the template parameter `NumberUnsignedType` which chooses the type to use. +## Template parameters + +`NumberUnsignedType` +: the type to store unsigned integers. It must be an **unsigned integral** type (`#!cpp std::is_integral`) with a + `#!cpp std::numeric_limits` specialization, and it must be able to represent the absolute value of every + [`number_integer_t`](number_integer_t.md) value. See + [Template Parameter Requirements](../../features/types/template_parameters.md#numberintegertype-and-numberunsignedtype). + ## Notes #### Default type diff --git a/docs/mkdocs/docs/api/basic_json/object_comparator_t.md b/docs/mkdocs/docs/api/basic_json/object_comparator_t.md index d41b98229..bbda0a0e7 100644 --- a/docs/mkdocs/docs/api/basic_json/object_comparator_t.md +++ b/docs/mkdocs/docs/api/basic_json/object_comparator_t.md @@ -30,3 +30,5 @@ and [`default_object_comparator_t`](default_object_comparator_t.md) otherwise. - Added in version 3.0.0. - Changed to be conditionally defined as `#!cpp typename object_t::key_compare` or `default_object_comparator_t` in version 3.11.0. +- Fixed the fallback to `default_object_comparator_t`, which previously failed to compile for object types without a + `key_compare` member type, in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json/object_t.md b/docs/mkdocs/docs/api/basic_json/object_t.md index de41b86e4..6ce393a1d 100644 --- a/docs/mkdocs/docs/api/basic_json/object_t.md +++ b/docs/mkdocs/docs/api/basic_json/object_t.md @@ -18,7 +18,11 @@ To store objects in C++, a type is defined by the template parameters described ## Template parameters `ObjectType` -: the container to store objects (e.g., `std::map` or `std::unordered_map`) +: the container to store objects. Its template parameters must have the same order and meaning as those of + `std::map`; in particular, the third parameter is a comparator. `#!cpp std::unordered_map`, whose third parameter + is a hash function, therefore needs an adapter -- see + [Template Parameter Requirements](../../features/types/template_parameters.md#objecttype) for the full list of + requirements, an adapter example, and the containers that are known to work. `StringType` : the type of the keys or names (e.g., `std::string`). The comparison function `std::less` is used to @@ -122,3 +126,4 @@ the object is silently converted as an array of key-value pairs, which is incorr ## Version history - Added in version 1.0.0. +- Allowed object types whose `erase(iterator)` returns `#!cpp void` in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json/string_t.md b/docs/mkdocs/docs/api/basic_json/string_t.md index 97c586f28..e8be0fe0e 100644 --- a/docs/mkdocs/docs/api/basic_json/string_t.md +++ b/docs/mkdocs/docs/api/basic_json/string_t.md @@ -23,6 +23,11 @@ JSON class into byte-sized characters during deserialization. `StringType`. To work with wide-character data, convert it to/from UTF-8 at the boundary instead -- see the FAQ's [wide string handling](../../home/faq.md#wide-string-handling) section for a conversion recipe. + Beyond the character type, the library expects a substantial part of the `#!cpp std::string` interface (contiguous + null-terminated `data()`, `substr()`, `find()`, `append()`, ...). See + [Template Parameter Requirements](../../features/types/template_parameters.md#stringtype) for the full list and + for the string types that are known to work. + ## Notes #### Default type @@ -78,3 +83,5 @@ and an example. ## Version history - Added in version 1.0.0. +- Removed the requirement that `string_t` be implicitly convertible from `#!cpp std::string`, which the BSON writer and + the UBJSON reader relied on, in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json/unflatten.md b/docs/mkdocs/docs/api/basic_json/unflatten.md index 1af243b58..ac88cd73e 100644 --- a/docs/mkdocs/docs/api/basic_json/unflatten.md +++ b/docs/mkdocs/docs/api/basic_json/unflatten.md @@ -37,7 +37,14 @@ Linear in the size of the JSON value. ## Notes Empty objects and arrays are flattened by [`flatten()`](flatten.md) to `#!json null` values and cannot unflattened to -their original type. Apart from this example, for a JSON value `j`, the following is always true: +their original type. + +A flattened array and a flattened object whose keys are array indices are indistinguishable, because both are +described by the same JSON pointers. A value is therefore restored as an array if and only if one of its keys is the +reference token `0`, and as an object otherwise: `#!json {"2": 1}` is restored unchanged, whereas `#!json {"0": 1}` is +restored as `#!json [1]`. This decision does not depend on the order in which the flattened object is iterated. + +Apart from these two cases, for a JSON value `j`, the following is always true: `#!cpp j == j.flatten().unflatten()`. ## Examples @@ -63,3 +70,4 @@ their original type. Apart from this example, for a JSON value `j`, the followin ## Version history - Added in version 2.0.0. +- Made the array/object decision independent of the object's iteration order in version 3.13.0. diff --git a/docs/mkdocs/docs/api/macros/json_throw_user.md b/docs/mkdocs/docs/api/macros/json_throw_user.md index b02918cf8..7f4fb6a8d 100644 --- a/docs/mkdocs/docs/api/macros/json_throw_user.md +++ b/docs/mkdocs/docs/api/macros/json_throw_user.md @@ -12,9 +12,11 @@ Controls how exceptions are handled by the library. 1. This macro overrides [`#!cpp catch`](https://en.cppreference.com/w/cpp/language/try_catch) calls inside the library. - The argument is the type of the exception to catch. As of version 3.8.0, the library only catches `std::out_of_range` - exceptions internally to rethrow them as [`json::out_of_range`](../../home/exceptions.md#out-of-range) exceptions. - The macro is always followed by a scope. + The argument is the type of the exception to catch. The library uses it in a single place: to swallow any exception + escaping the parent-pointer check that [`JSON_DIAGNOSTICS`](json_diagnostics.md) adds to the class invariant. The + places where the library catches its own [`json::out_of_range`](../../home/exceptions.md#out-of-range) exceptions + use `JSON_INTERNAL_CATCH` instead, which `JSON_CATCH_USER` also overrides unless `JSON_INTERNAL_CATCH_USER` is + defined. The macro is always followed by a scope. 2. This macro overrides `#!cpp throw` calls inside the library. The argument is the exception to be thrown. Note that `JSON_THROW_USER` should leave the current scope (e.g., by throwing or aborting), as continuing after it may yield undefined behavior. diff --git a/docs/mkdocs/docs/examples/custom_array_type.cpp b/docs/mkdocs/docs/examples/custom_array_type.cpp new file mode 100644 index 000000000..63651cadf --- /dev/null +++ b/docs/mkdocs/docs/examples/custom_array_type.cpp @@ -0,0 +1,19 @@ +#include +#include + +#include + +#include "custom_array_type.hpp" + +using custom_json = nlohmann::basic_json; + +int main() +{ + custom_json j = custom_json::array(); + j.push_back(1); + j.push_back(2); + j.push_back(3); + + std::cout << j.dump() << std::endl; + std::cout << std::boolalpha << (custom_json::parse(j.dump()) == j) << std::endl; +} diff --git a/docs/mkdocs/docs/examples/custom_array_type.hpp b/docs/mkdocs/docs/examples/custom_array_type.hpp new file mode 100644 index 000000000..750d52a81 --- /dev/null +++ b/docs/mkdocs/docs/examples/custom_array_type.hpp @@ -0,0 +1,152 @@ +#pragma once + +#include +#include +#include + +// A minimal, self-contained ArrayType built around a private std::vector. +// See https://json.nlohmann.me/features/types/template_parameters/#arraytype +template> +class custom_array_type +{ + using vector_t = std::vector; + vector_t data_; + + public: + using value_type = typename vector_t::value_type; + using size_type = typename vector_t::size_type; + using iterator = typename vector_t::iterator; + using const_iterator = typename vector_t::const_iterator; + + custom_array_type() = default; + custom_array_type(const custom_array_type&) = default; + custom_array_type(custom_array_type&&) = default; + custom_array_type& operator=(const custom_array_type&) = default; + custom_array_type& operator=(custom_array_type&&) = default; + + template + custom_array_type(InputIt first, InputIt last) : data_(first, last) {} + + custom_array_type(size_type count, const T& value) : data_(count, value) {} + + iterator begin() + { + return data_.begin(); + } + iterator end() + { + return data_.end(); + } + const_iterator begin() const + { + return data_.begin(); + } + const_iterator end() const + { + return data_.end(); + } + const_iterator cbegin() const + { + return data_.cbegin(); + } + const_iterator cend() const + { + return data_.cend(); + } + + bool empty() const + { + return data_.empty(); + } + size_type size() const + { + return data_.size(); + } + size_type max_size() const + { + return data_.max_size(); + } + void clear() + { + data_.clear(); + } + void resize(size_type n) + { + data_.resize(n); + } + + T& operator[](size_type pos) + { + return data_[pos]; + } + const T& operator[](size_type pos) const + { + return data_[pos]; + } + + T& back() + { + return data_.back(); + } + const T& back() const + { + return data_.back(); + } + + void push_back(const T& value) + { + data_.push_back(value); + } + void push_back(T&& value) + { + data_.push_back(std::move(value)); + } + + template + void emplace_back(Args&& ... args) + { + data_.emplace_back(std::forward(args)...); + } + + void pop_back() + { + data_.pop_back(); + } + + iterator insert(const_iterator pos, const T& value) + { + return data_.insert(pos, value); + } + iterator insert(const_iterator pos, size_type count, const T& value) + { + return data_.insert(pos, count, value); + } + template + iterator insert(const_iterator pos, InputIt first, InputIt last) + { + return data_.insert(pos, first, last); + } + + iterator erase(const_iterator pos) + { + return data_.erase(pos); + } + iterator erase(const_iterator first, const_iterator last) + { + return data_.erase(first, last); + } + + void swap(custom_array_type& other) + { + data_.swap(other.data_); + } + + friend bool operator==(const custom_array_type& lhs, const custom_array_type& rhs) + { + return lhs.data_ == rhs.data_; + } + friend bool operator<(const custom_array_type& lhs, const custom_array_type& rhs) + { + return lhs.data_ < rhs.data_; + } +}; diff --git a/docs/mkdocs/docs/examples/custom_array_type.output b/docs/mkdocs/docs/examples/custom_array_type.output new file mode 100644 index 000000000..f5ceaa555 --- /dev/null +++ b/docs/mkdocs/docs/examples/custom_array_type.output @@ -0,0 +1,2 @@ +[1,2,3] +true diff --git a/docs/mkdocs/docs/examples/custom_binary_type.cpp b/docs/mkdocs/docs/examples/custom_binary_type.cpp new file mode 100644 index 000000000..cea34ea38 --- /dev/null +++ b/docs/mkdocs/docs/examples/custom_binary_type.cpp @@ -0,0 +1,21 @@ +#include +#include +#include +#include +#include + +#include + +#include "custom_binary_type.hpp" + +using custom_json = nlohmann::basic_json; + +int main() +{ + const auto j = custom_json::binary({0x01, 0x02, 0x03}); + + std::cout << j.dump() << std::endl; + std::cout << std::boolalpha << (custom_json::from_cbor(custom_json::to_cbor(j)) == j) << std::endl; +} diff --git a/docs/mkdocs/docs/examples/custom_binary_type.hpp b/docs/mkdocs/docs/examples/custom_binary_type.hpp new file mode 100644 index 000000000..77b4ad49a --- /dev/null +++ b/docs/mkdocs/docs/examples/custom_binary_type.hpp @@ -0,0 +1,112 @@ +#pragma once + +#include +#include +#include + +// A minimal, self-contained BinaryType built around a private std::vector. +// See https://json.nlohmann.me/features/types/template_parameters/#binarytype +class custom_binary_type +{ + using vector_t = std::vector; + vector_t data_; + + public: + using value_type = vector_t::value_type; + using size_type = vector_t::size_type; + using iterator = vector_t::iterator; + using const_iterator = vector_t::const_iterator; + + custom_binary_type() = default; + custom_binary_type(const custom_binary_type&) = default; + custom_binary_type(custom_binary_type&&) = default; + custom_binary_type& operator=(const custom_binary_type&) = default; + custom_binary_type& operator=(custom_binary_type&&) = default; + + template + custom_binary_type(InputIt first, InputIt last) : data_(first, last) {} + + // so basic_json::binary({0x01, 0x02}) can build one directly + custom_binary_type(std::initializer_list init) : data_(init) {} + + size_type size() const + { + return data_.size(); + } + bool empty() const + { + return data_.empty(); + } + void clear() + { + data_.clear(); + } + void resize(size_type n) + { + data_.resize(n); + } + + // read-only is enough: the writers only ever read from a binary value + const std::uint8_t* data() const + { + return data_.data(); + } + + std::uint8_t& operator[](size_type pos) + { + return data_[pos]; + } + std::uint8_t operator[](size_type pos) const + { + return data_[pos]; + } + + std::uint8_t& back() + { + return data_.back(); + } + std::uint8_t back() const + { + return data_.back(); + } + + iterator begin() + { + return data_.begin(); + } + iterator end() + { + return data_.end(); + } + const_iterator begin() const + { + return data_.begin(); + } + const_iterator end() const + { + return data_.end(); + } + const_iterator cbegin() const + { + return data_.cbegin(); + } + const_iterator cend() const + { + return data_.cend(); + } + + template + iterator insert(const_iterator pos, InputIt first, InputIt last) + { + return data_.insert(pos, first, last); + } + + friend bool operator==(const custom_binary_type& lhs, const custom_binary_type& rhs) + { + return lhs.data_ == rhs.data_; + } + friend bool operator<(const custom_binary_type& lhs, const custom_binary_type& rhs) + { + return lhs.data_ < rhs.data_; + } +}; diff --git a/docs/mkdocs/docs/examples/custom_binary_type.output b/docs/mkdocs/docs/examples/custom_binary_type.output new file mode 100644 index 000000000..b4814d6ed --- /dev/null +++ b/docs/mkdocs/docs/examples/custom_binary_type.output @@ -0,0 +1,2 @@ +{"bytes":[1,2,3],"subtype":null} +true diff --git a/docs/mkdocs/docs/examples/custom_object_type.cpp b/docs/mkdocs/docs/examples/custom_object_type.cpp new file mode 100644 index 000000000..d4f99ccf0 --- /dev/null +++ b/docs/mkdocs/docs/examples/custom_object_type.cpp @@ -0,0 +1,26 @@ +#include +#include +#include + +#include + +#include "custom_object_type.hpp" + +using custom_json = nlohmann::basic_json; + +int main() +{ + custom_json j; + j["pi"] = 3.141; + j["happy"] = true; + j["list"] = {1, 2, 3}; + + std::cout << j.dump(2) << std::endl; + std::cout << std::boolalpha << (custom_json::parse(j.dump()) == j) << std::endl; + + // custom_object_type has no key_compare member, so object_comparator_t + // falls back to its default + std::cout << std::boolalpha + << std::is_same::value + << std::endl; +} diff --git a/docs/mkdocs/docs/examples/custom_object_type.hpp b/docs/mkdocs/docs/examples/custom_object_type.hpp new file mode 100644 index 000000000..da715ed4e --- /dev/null +++ b/docs/mkdocs/docs/examples/custom_object_type.hpp @@ -0,0 +1,144 @@ +#pragma once + +#include +#include + +// A minimal, self-contained ObjectType built around a private std::map. +// key_compare is deliberately not exposed: when an ObjectType has no +// key_compare member, the library falls back to its own default comparator. +// See https://json.nlohmann.me/features/types/template_parameters/#objecttype +template +class custom_object_type +{ + using map_t = std::map; + map_t data_; + + public: + using key_type = typename map_t::key_type; + using mapped_type = typename map_t::mapped_type; + using value_type = typename map_t::value_type; + using size_type = typename map_t::size_type; + using iterator = typename map_t::iterator; + using const_iterator = typename map_t::const_iterator; + + custom_object_type() = default; + custom_object_type(const custom_object_type&) = default; + custom_object_type(custom_object_type&&) = default; + custom_object_type& operator=(const custom_object_type&) = default; + custom_object_type& operator=(custom_object_type&&) = default; + + template + custom_object_type(InputIt first, InputIt last) : data_(first, last) {} + + iterator begin() + { + return data_.begin(); + } + iterator end() + { + return data_.end(); + } + const_iterator begin() const + { + return data_.begin(); + } + const_iterator end() const + { + return data_.end(); + } + const_iterator cbegin() const + { + return data_.cbegin(); + } + const_iterator cend() const + { + return data_.cend(); + } + + bool empty() const + { + return data_.empty(); + } + size_type size() const + { + return data_.size(); + } + size_type max_size() const + { + return data_.max_size(); + } + void clear() + { + data_.clear(); + } + + iterator find(const key_type& key) + { + return data_.find(key); + } + const_iterator find(const key_type& key) const + { + return data_.find(key); + } + size_type count(const key_type& key) const + { + return data_.count(key); + } + + std::pair emplace(const key_type& key, const mapped_type& value) + { + return data_.emplace(key, value); + } + + std::pair insert(const value_type& value) + { + return data_.insert(value); + } + + template + void insert(InputIt first, InputIt last) + { + data_.insert(first, last); + } + + mapped_type& operator[](const key_type& key) + { + return data_[key]; + } + + mapped_type& at(const key_type& key) + { + return data_.at(key); + } + const mapped_type& at(const key_type& key) const + { + return data_.at(key); + } + + iterator erase(iterator pos) + { + return data_.erase(pos); + } + iterator erase(iterator first, iterator last) + { + return data_.erase(first, last); + } + size_type erase(const key_type& key) + { + return data_.erase(key); + } + + void swap(custom_object_type& other) + { + data_.swap(other.data_); + } + + friend bool operator==(const custom_object_type& lhs, const custom_object_type& rhs) + { + return lhs.data_ == rhs.data_; + } + friend bool operator<(const custom_object_type& lhs, const custom_object_type& rhs) + { + return lhs.data_ < rhs.data_; + } +}; diff --git a/docs/mkdocs/docs/examples/custom_object_type.output b/docs/mkdocs/docs/examples/custom_object_type.output new file mode 100644 index 000000000..48b3e9630 --- /dev/null +++ b/docs/mkdocs/docs/examples/custom_object_type.output @@ -0,0 +1,11 @@ +{ + "happy": true, + "list": [ + 1, + 2, + 3 + ], + "pi": 3.141 +} +true +true diff --git a/docs/mkdocs/docs/examples/custom_string_type.cpp b/docs/mkdocs/docs/examples/custom_string_type.cpp new file mode 100644 index 000000000..63b798fc6 --- /dev/null +++ b/docs/mkdocs/docs/examples/custom_string_type.cpp @@ -0,0 +1,20 @@ +#include +#include +#include + +#include + +#include "custom_string_type.hpp" + +using custom_json = nlohmann::basic_json; + +int main() +{ + custom_json j; + j["pi"] = 3.141; + j["happy"] = true; + j["list"] = {1, 2, 3}; + + std::cout << j.dump(2) << std::endl; + std::cout << std::boolalpha << (custom_json::parse(j.dump()) == j) << std::endl; +} diff --git a/docs/mkdocs/docs/examples/custom_string_type.hpp b/docs/mkdocs/docs/examples/custom_string_type.hpp new file mode 100644 index 000000000..ec48501fc --- /dev/null +++ b/docs/mkdocs/docs/examples/custom_string_type.hpp @@ -0,0 +1,134 @@ +#pragma once + +#include +#include + +// A minimal, self-contained StringType built around a private std::string. +// Wraps rather than inherits, so it exposes exactly what the library needs +// and nothing more of std::string's interface. +// +// Covers the "Always required" members, the extras needed for the binary +// formats, and the extras needed for JSON Pointer / flatten / unflatten / +// diff. Extending it further (e.g. for std::hash or to_bson) is +// a matter of adding the extra members listed in the "Required for other +// functionality" table. +// +// See https://json.nlohmann.me/features/types/template_parameters/#stringtype +class custom_string_type +{ + std::string data_; + + public: + using value_type = char; + using size_type = std::string::size_type; + using iterator = std::string::iterator; + using const_iterator = std::string::const_iterator; + + static constexpr size_type npos = std::string::npos; + + custom_string_type() = default; + custom_string_type(const custom_string_type&) = default; + custom_string_type(custom_string_type&&) = default; + custom_string_type& operator=(const custom_string_type&) = default; + custom_string_type& operator=(custom_string_type&&) = default; + + // not explicit: the library relies on being able to hand it a string literal + custom_string_type(const char* s) : data_(s) {} + custom_string_type(const char* s, size_type count) : data_(s, count) {} + custom_string_type(size_type count, char ch) : data_(count, ch) {} + + size_type size() const + { + return data_.size(); + } + bool empty() const + { + return data_.empty(); + } + void clear() + { + data_.clear(); + } + void resize(size_type n) + { + data_.resize(n); + } + void resize(size_type n, char c) + { + data_.resize(n, c); + } + void reserve(size_type n) + { + data_.reserve(n); + } + + // must stay null-terminated -- the parser hands this to std::strtoull & + // friends; std::string::data() has guaranteed that since C++11 + const char* data() const + { + return data_.data(); + } + + void push_back(char c) + { + data_.push_back(c); + } + + char& operator[](size_type pos) + { + return data_[pos]; + } + char operator[](size_type pos) const + { + return data_[pos]; + } + + custom_string_type& append(const char* s, size_type count) + { + data_.append(s, count); + return *this; + } + custom_string_type& append(const custom_string_type& other) + { + data_.append(other.data_); + return *this; + } + + size_type find_first_of(char c, size_type pos = 0) const + { + return data_.find_first_of(c, pos); + } + + iterator begin() + { + return data_.begin(); + } + iterator end() + { + return data_.end(); + } + const_iterator begin() const + { + return data_.begin(); + } + const_iterator end() const + { + return data_.end(); + } + + friend bool operator==(const custom_string_type& lhs, const custom_string_type& rhs) + { + return lhs.data_ == rhs.data_; + } + friend bool operator<(const custom_string_type& lhs, const custom_string_type& rhs) + { + return lhs.data_ < rhs.data_; + } + + // not required by the library itself, but dump() returns a custom_string_type + // and this makes `std::cout << j.dump()` work as expected + friend std::ostream& operator<<(std::ostream& os, const custom_string_type& s) + { + return os << s.data_; + } +}; diff --git a/docs/mkdocs/docs/examples/custom_string_type.output b/docs/mkdocs/docs/examples/custom_string_type.output new file mode 100644 index 000000000..d792e2a72 --- /dev/null +++ b/docs/mkdocs/docs/examples/custom_string_type.output @@ -0,0 +1,10 @@ +{ + "happy": true, + "list": [ + 1, + 2, + 3 + ], + "pi": 3.141 +} +true diff --git a/docs/mkdocs/docs/features/object_order.md b/docs/mkdocs/docs/features/object_order.md index f62474efd..200913fd2 100644 --- a/docs/mkdocs/docs/features/object_order.md +++ b/docs/mkdocs/docs/features/object_order.md @@ -51,7 +51,11 @@ If you do want to preserve the **insertion order**, you can use the type [`nlohm --8<-- "examples/ordered_json.output" ``` -Alternatively, you can use a more sophisticated ordered map like [`tsl::ordered_map`](https://github.com/Tessil/ordered-map) ([integration](https://github.com/nlohmann/json/issues/546#issuecomment-304447518)) or [`nlohmann::fifo_map`](https://github.com/nlohmann/fifo_map) ([integration](https://github.com/nlohmann/json/issues/485#issuecomment-333652309)). +Alternatively, [`nlohmann::fifo_map`](https://github.com/nlohmann/fifo_map) also preserves the insertion order and, unlike [`ordered_map`](../api/ordered_map.md), keeps a lookup index, so it does not have the quadratic cost described below. It is used through a small adapter ([integration](https://github.com/nlohmann/json/issues/485#issuecomment-333652309)). + +If the order does not matter and you only want faster lookup, `boost::unordered_flat_map`, `absl::flat_hash_map`, `absl::node_hash_map`, and several other hash maps work through an adapter that restores the template argument order `basic_json` expects; see [Template Parameter Requirements](types/template_parameters.md#objecttype). Note these are *unordered*, not insertion-ordered. + +[`tsl::ordered_map`](https://github.com/Tessil/ordered-map) cannot be used: its iterators expose the mapped value as `const`, while `basic_json` needs to modify it in place. The [`ordered_map`](../api/ordered_map.md) behind `nlohmann::ordered_json` is deliberately minimal and has no lookup index, so every key access is a linear scan and building an object of `n` keys costs O(n²). This is unnoticeable at diff --git a/docs/mkdocs/docs/features/types/index.md b/docs/mkdocs/docs/features/types/index.md index 5990ac708..e6078b825 100644 --- a/docs/mkdocs/docs/features/types/index.md +++ b/docs/mkdocs/docs/features/types/index.md @@ -79,7 +79,8 @@ template< class NumberFloatType = double, template class AllocatorType = std::allocator, template class JSONSerializer = adl_serializer, - class BinaryType = std::vector + class BinaryType = std::vector, + class CustomBaseClass = void > class basic_json; ``` @@ -106,6 +107,10 @@ using number_float_t = NumberFloatType; using binary_t = nlohmann::byte_container_with_subtype; ``` +Not every type can be passed for these template arguments: the library uses the resulting types in ways that imply a +number of requirements, for instance that `StringType` is `char`-based or that `ArrayType` is vector-like. These +requirements are collected in [Template Parameter Requirements](template_parameters.md). + ## Objects diff --git a/docs/mkdocs/docs/features/types/template_parameters.md b/docs/mkdocs/docs/features/types/template_parameters.md new file mode 100644 index 000000000..760972dff --- /dev/null +++ b/docs/mkdocs/docs/features/types/template_parameters.md @@ -0,0 +1,747 @@ +# Template Parameter Requirements + +Class [`basic_json`](../../api/basic_json/index.md) is configurable through eleven template parameters. The library +never formally states what a type passed for one of these parameters has to provide -- the requirements are implied by +the way the library uses the resulting [`object_t`](../../api/basic_json/object_t.md), +[`array_t`](../../api/basic_json/array_t.md), [`string_t`](../../api/basic_json/string_t.md), etc. This page collects +these requirements so they do not have to be discovered by trial and error. Each section lists the concrete types +that are known to work for that parameter and the ones that do not, checked against Boost 1.83, Abseil 20250127.0, +Folly, EASTL 3.21, `ankerl::unordered_dense`, `phmap`, `gtl`, `robin_hood`, `tsl::ordered_map`, and Qt 6. + +## How to read this page + +Requirements are split into two groups: + +- **Always required** -- needed to instantiate `basic_json` at all, or needed by functions that virtually every program + uses (construction, element access, [`dump`](../../api/basic_json/dump.md)). +- **Required for ...** -- only needed when a particular part of the API is instantiated. Member function templates are + only instantiated when they are used, so a type may be perfectly usable even though it does not satisfy these + requirements, as long as the corresponding functions are never called. + +!!! warning "Requirements are not checked" + + Three requirements are checked with a `#!cpp static_assert`: the array iterator category, the width of + [`BinaryType`](#binarytype)'s `value_type`, and [`NumberUnsignedType`](#numberintegertype-and-numberunsignedtype) + being at least as wide as [`NumberIntegerType`](#numberintegertype-and-numberunsignedtype). The rest are not + diagnosed with dedicated error messages, and violating most of them results in a compiler error somewhere inside + the library. Four violations are not caught at compile time at all: + + - A [`StringType`](#stringtype) whose `data()` is not null-terminated compiles and silently misparses numbers, + because the lexer hands the buffer to `#!cpp std::strtoull`/`#!cpp std::strtoll`/`#!cpp std::strtod`. + - A stateful [`AllocatorType`](#allocatortype) compiles and silently ignores its state: allocation, deallocation, + and [`get_allocator()`](../../api/basic_json/get_allocator.md) each use a different default-constructed instance. + - The two [cross-specialization conversions](#cross-specialization-conversions) below. These abort on an assertion + in a normal build, and only fail silently under `#!cpp NDEBUG`. + +## Overview + +| Template parameter | Default | Notable substitutes | +|-------------------------------------------------------------------|-----------------------------------|-----------------------------------------------------------------------| +| [`ObjectType`](#objecttype) | `std::map` | [`nlohmann::ordered_map`](../../api/ordered_map.md), Abseil hash maps | +| [`ArrayType`](#arraytype) | `std::vector` | `#!cpp std::deque` | +| [`StringType`](#stringtype) | `std::string` | `std::string`-like types over `char` | +| [`BooleanType`](#booleantype) | `bool` | none worth using | +| [`NumberIntegerType`](#numberintegertype-and-numberunsignedtype) | `std::int64_t` | any signed integer type | +| [`NumberUnsignedType`](#numberintegertype-and-numberunsignedtype) | `std::uint64_t` | any unsigned integer type at least as wide as `NumberIntegerType` | +| [`NumberFloatType`](#numberfloattype) | `double` | `float` (`long double`: no binary formats) | +| [`AllocatorType`](#allocatortype) | `std::allocator` | stateless allocators | +| [`JSONSerializer`](#jsonserializer) | `adl_serializer` | serializers with the same interface | +| [`BinaryType`](#binarytype) | `#!cpp std::vector` | `#!cpp std::vector` | +| [`CustomBaseClass`](#custombaseclass) | `void` | any default-constructible class | + +!!! warning "Third-party containers and incomplete types" + + `object_t` is instantiated inside the definition of `basic_json` -- it is probed for a `key_compare` member to + form [`object_comparator_t`](../../api/basic_json/object_comparator_t.md) -- i.e. while `basic_json` is still an + incomplete type. `#!cpp std::map` is required by the standard to support incomplete mapped types; most + third-party maps are not, and inspecting the mapped type at class scope (for instance with + `#!cpp std::is_trivially_move_assignable`) makes them unusable as `ObjectType`, no matter how their template + arguments are adapted. This rules out `absl::btree_map`, `phmap::btree_map`, `gtl::btree_map`, + `robin_hood::unordered_node_map`, `folly::F14FastMap`, and `eastl::hash_map`. + + `array_t` is only *named* in the class definition and is not instantiated until `basic_json` is complete, so an + `ArrayType` that inspects its value type at class scope is generally fine -- `boost::container::small_vector` and + `static_vector` both reject incomplete value types yet work here. `absl::InlinedVector` is the exception: the + `#!cpp std::is_trivially_move_assignable` it evaluates while instantiating itself re-enters the + library's own trait machinery mid-instantiation. + +!!! note "Folly requires C++20" + + Folly's headers use `#!cpp consteval` and `#!cpp std::type_identity`, so any `basic_json` specialization that + names a Folly type has to be compiled as C++20 or later, whatever the rest of the library supports. + +## `ObjectType` + +`ObjectType` is instantiated as + +```cpp +using object_t = ObjectType>>; // allocator_type +``` + +i.e., the template arguments follow the order and meaning of `std::map`. + +### Always required + +- The template must be usable with **four** type arguments in the order shown above. The third argument is a + **comparator**; containers that expect something else in this position (e.g., a hash function) need an alias template + or wrapper -- see [Notes](#notes). +- An optional member type `key_compare`. If it is present it becomes + [`object_comparator_t`](../../api/basic_json/object_comparator_t.md); otherwise + [`default_object_comparator_t`](../../api/basic_json/default_object_comparator_t.md) is used. +- Member types `key_type`, `mapped_type`, `value_type`, and `iterator`. +- `value_type` must behave like `#!cpp std::pair`; the library accesses `.first` and + `.second` on it. +- `iterator` must be default-constructible and satisfy + [LegacyBidirectionalIterator](https://en.cppreference.com/w/cpp/named_req/BidirectionalIterator). The type returned + by `cbegin()`/`cend()` must satisfy the same requirements. +- Constructors: default, copy, move, and from an iterator range `(first, last)`. +- Member functions `begin()`, `end()`, `cbegin()`, `cend()`, `empty()`, `size()`, `max_size()`, `clear()`, + `find(key)`, `count(key)`, `emplace(key, value)`, `insert(value_type)`, `insert(first, last)`, `operator[](key)`, + `erase(iterator)`, and `erase(first, last)`. `erase(iterator)` may return the following iterator or `#!cpp void`; + in the latter case the library computes the successor itself, before erasing. +- `erase(key)` is **optional**: if the container does not provide one, the library falls back to `find(key)` followed + by `erase(iterator)`. +- `at(key)` is required only by [`to_ubjson`](../../api/basic_json/to_ubjson.md) and + [`to_bjdata`](../../api/basic_json/to_bjdata.md), but every container tried here provides it. +- `emplace` and `insert(value_type)` must return `#!cpp std::pair` and must have **unique-key** + semantics; multimaps cannot be used. +- The type must be swappable (via `std::swap` or an ADL `swap`). +- The comparison operators `==` and `<`; `!=`, `<=`, `>`, and `>=` are derived from them. Where the library uses + three-way comparison (C++20), `==` and `<=>` are required **instead** -- the six two-way operators do not satisfy + it. They implement [`basic_json`'s comparison operators](../../api/basic_json/operator_eq.md). + +### Required for heterogeneous key lookup + +The overloads of [`at`](../../api/basic_json/at.md), [`operator[]`](../../api/basic_json/operator%5B%5D.md), +[`find`](../../api/basic_json/find.md), [`contains`](../../api/basic_json/contains.md), +[`count`](../../api/basic_json/count.md), [`erase`](../../api/basic_json/erase.md), and +[`value`](../../api/basic_json/value.md) that accept a key type other than `object_t::key_type` require + +- a **transparent** comparator, i.e. [`object_comparator_t`](../../api/basic_json/object_comparator_t.md) has a member + type `is_transparent` (this is why the default comparator is `#!cpp std::less<>` since C++14), and +- corresponding heterogeneous `find`, `count`, `erase`, and `operator[]` overloads on the container. + +### Notes + +#### `std::unordered_map` needs an adapter + +`#!cpp std::unordered_map` cannot be passed directly: its third template parameter is a hash function, but +`basic_json` passes a comparator in that position. An alias template or wrapper that restores the expected argument +order makes it usable: + +```cpp +template +struct unordered_map_object + : std::unordered_map, std::equal_to, Allocator> +{ + using base_t = std::unordered_map, std::equal_to, Allocator>; + using base_t::base_t; +}; + +using unordered_json = nlohmann::basic_json; +``` + +Whether `#!cpp std::unordered_map` can be instantiated at all depends on the standard library: `object_t` is formed +while `basic_json` is still incomplete (see the warning above), and libstdc++ 9 needs the size of the mapped type to +instantiate the hash map's node type, so the adapter does not compile there. Newer libstdc++ versions, and the hash +maps listed below, do not have that problem. + +The adapter above works verbatim for Abseil's, Boost's, `phmap`'s and `gtl`'s hash maps, which all place the hash +function third and take a `#!cpp std::pair` allocator fifth. Two need a different adapter: + +- `ankerl::unordered_dense` expects an allocator over `#!cpp std::pair` (non-const key), so the allocator has + to be rebound to that or dropped. +- `robin_hood`'s fifth parameter is the non-type `MaxLoadFactor100`, so its adapter must drop the allocator entirely. + +None of these hash maps defines `key_compare`, so all of them additionally rely on `object_comparator_t` falling back +to [`default_object_comparator_t`](../../api/basic_json/default_object_comparator_t.md); see +[`object_comparator_t`](../../api/basic_json/object_comparator_t.md). + +#### Abseil hash maps + +`absl::flat_hash_map` and `absl::node_hash_map` tolerate an incomplete value type, but they take a hash function as +their third template argument. The same adapter as for `#!cpp std::unordered_map` makes them usable: + +```cpp +template +struct flat_hash_object + : absl::flat_hash_map, std::equal_to, Allocator> +{ + using base_t = absl::flat_hash_map, std::equal_to, Allocator>; + using base_t::base_t; +}; + +using flat_hash_json = nlohmann::basic_json; +``` + +`absl::node_hash_map` keeps references to the mapped values valid across insertions; `absl::flat_hash_map` does not, +which makes it behave like [`ordered_json`](../../api/ordered_json.md) with respect to +[iterator invalidation](../../api/basic_json/index.md#iterator-invalidation). Both expose a `capacity()` member +function, so [`JSON_DIAGNOSTICS`](../../api/macros/json_diagnostics.md) treats them conservatively and keeps the +parent pointers correct either way. + +#### Iteration order + +The library never relies on the container's iteration order for correctness; it does determine the order in which +object keys are serialized by [`dump`](../../api/basic_json/dump.md) and visited by +[`items`](../../api/basic_json/items.md). See [Object Order](../object_order.md). + +#### `capacity()` marks a container as insertion-ordered + +With [`JSON_DIAGNOSTICS`](../../api/macros/json_diagnostics.md) enabled, the library detects insertion-ordered maps by +probing for a `capacity()` member function (`nlohmann::ordered_map` inherits it from `std::vector`) and refreshes all +parent pointers after every insertion. An `ObjectType` that happens to have a `capacity()` member is therefore treated +conservatively -- this is correct, but slower. + +#### Key order and duplicate keys + +The library does not sort or de-duplicate keys itself; the behavior described in +[`object_t`](../../api/basic_json/object_t.md) is entirely the behavior of the chosen container. + +!!! tip "Reference implementation" + + `docs/mkdocs/docs/examples/custom_object_type.hpp` wraps a private `#!cpp std::map` and satisfies every + requirement above. It does not define `key_compare`, so `object_comparator_t` falls back to + [`default_object_comparator_t`](../../api/basic_json/default_object_comparator_t.md) -- a good starting point for + a custom `ObjectType`. + + ```cpp + --8<-- "examples/custom_object_type.hpp" + ``` + +??? example "Compiling and using it" + + ```cpp + --8<-- "examples/custom_object_type.cpp" + ``` + + Output: + + ```json + --8<-- "examples/custom_object_type.output" + ``` + +### Compatible containers + +| Container | Notes | +|----------------------------------------------------------------------------------|-------------------------------------------------------------------------------| +| `#!cpp std::map` (default) | | +| [`nlohmann::ordered_map`](../../api/ordered_map.md) | used by [`ordered_json`](../../api/ordered_json.md); keeps insertion order | +| [`nlohmann::fifo_map`](https://github.com/nlohmann/fifo_map) | keeps insertion order; adapter puts `fifo_map_compare` in the comparator slot | +| `boost::container::map`, `boost::container::flat_map` | no adapter needed | +| `#!cpp std::unordered_map` | through the adapter above; not with libstdc++ 9, see the note | +| `boost::unordered_map`, `boost::unordered_flat_map`, `boost::unordered_node_map` | through the adapter above | +| `absl::flat_hash_map`, `absl::node_hash_map` | through the adapter above; `flat_hash_map` moves mapped values on rehash | +| `phmap::flat_hash_map`, `phmap::node_hash_map`, `gtl::flat_hash_map` | through the adapter above | +| `ankerl::unordered_dense::map` and `segmented_map` | adapter must rebind or drop the allocator | +| `robin_hood::unordered_flat_map` | adapter must drop the allocator | +| `folly::F14NodeMap` | through the adapter above; requires C++20, see the note above | +| `folly::sorted_vector_map` | alias must drop the allocator, whose value type it disagrees on | + +### Containers that cannot be used + +| Container | Reason | +|--------------------------------------------------------------------------|---------------------------------------------------------------------------------------------------------------------| +| `absl::btree_map`, `phmap::btree_map`, `gtl::btree_map` | require a complete mapped type | +| `robin_hood::unordered_node_map`, `folly::F14FastMap`, `eastl::hash_map` | require a complete mapped type | +| `eastl::map` | EASTL iterators do not work with `#!cpp std::iterator_traits` | +| `tsl::ordered_map` | its iterators expose the mapped value as `#!cpp const` | +| `QMap` | no `value_type` member type | +| `QHash` | its `value_type` is the mapped type rather than a key/value pair, and its iterators dereference to the mapped value | +| `#!cpp std::multimap`, `#!cpp std::unordered_multimap` | `emplace` does not return `#!cpp std::pair` | + +## `ArrayType` + +`ArrayType` is instantiated as + +```cpp +using array_t = ArrayType>; +``` + +### Always required + +- The template must be usable with **two** type arguments (value type and allocator). +- Member types `value_type` and `iterator`. +- Constructors: default, copy, and move; and from an iterator range `(first, last)`. +- Member functions `begin()`, `end()`, `cbegin()`, `cend()`, `empty()`, `size()`, `max_size()`, `clear()`, + `operator[](size_type)`, `back()`, `push_back()`, `emplace_back()`, `pop_back()`, `resize()`, + `insert()` (single element, count, and range), `erase(pos)`, and `erase(first, last)`. + `basic_json::insert(pos, initializer_list)` goes through the range overload, so no initializer-list `insert` is + needed. `at(size_type)` is **not** required: [`basic_json::at(size_type)`](../../api/basic_json/at.md) checks the + index itself and then uses `operator[]`. +- `iterator` must be default-constructible, and it as well as the type returned by `cbegin()`/`cend()` must satisfy + [LegacyRandomAccessIterator](https://en.cppreference.com/w/cpp/named_req/RandomAccessIterator). + A `#!cpp static_assert` only checks for + [LegacyBidirectionalIterator](https://en.cppreference.com/w/cpp/named_req/BidirectionalIterator), but + [`dump`](../../api/basic_json/dump.md) (`cend() - 1`), + [`erase(idx)`](../../api/basic_json/erase.md) (`begin() + idx`), and the random-access operations of + [`basic_json::iterator`](../../api/basic_json/begin.md) require random access. +- The comparison operators, as for [`ObjectType`](#objecttype): `==` and `<`, or `==` and `<=>` under C++20. + +### Required for individual functions + +- A member type `value_type`, for [`to_bson`](../../api/basic_json/to_bson.md) of an array. +- A constructor from `(count, value)`, for + [`basic_json(size_type, const basic_json&)`](../../api/basic_json/basic_json.md). +- Swappability, via `#!cpp std::swap` or an ADL `swap`, for [`swap(array_t&)`](../../api/basic_json/swap.md). + +!!! note "`capacity()` is optional" + + With [`JSON_DIAGNOSTICS`](../../api/macros/json_diagnostics.md) enabled, the library reads `array_t::capacity()` + to find out whether adding an element reallocated the array and moved its elements, which would invalidate the + parent pointers. An array type without a `capacity()` member function is handled conservatively: the parent + pointers of all elements are refreshed after every insertion, which makes adding *n* elements cost O(*n*²). Only + diagnostics builds pay this; without them `capacity()` is never called. + +!!! tip "Reference implementation" + + `docs/mkdocs/docs/examples/custom_array_type.hpp` wraps a private `#!cpp std::vector` and satisfies every + requirement above -- a good starting point for a custom `ArrayType`. + + ```cpp + --8<-- "examples/custom_array_type.hpp" + ``` + +??? example "Compiling and using it" + + ```cpp + --8<-- "examples/custom_array_type.cpp" + ``` + + Output: + + ```json + --8<-- "examples/custom_array_type.output" + ``` + +### Compatible containers + +| Container | Notes | +|---------------------------------------------------------|-------------------------------------------------------------------------------------------| +| `#!cpp std::vector` (default) | | +| `#!cpp std::deque` | references survive appends, but not insertions elsewhere; see the `capacity()` note above | +| `#!cpp std::pmr::vector` | through an alias, as the allocator comes from `AllocatorType` instead | +| `boost::container::vector`, `deque`, `devector` | | +| `boost::container::stable_vector` | the only one tried that keeps references valid across *every* insertion | +| `boost::container::small_vector`, `folly::small_vector` | through an alias that fixes the inline capacity | +| `boost::container::static_vector` | through the same kind of alias, for arrays that stay within the fixed capacity | +| `folly::fbvector` | requires C++20, see the note above | + +### Containers that cannot be used + +| Container | Reason | +|-------------------------------------|-----------------------------------------------------------------------------------------------| +| `#!cpp std::list` | no `operator[]`, and no random-access iterators | +| `eastl::vector`, `QList`, `QVector` | no `max_size()`; they handle the incomplete value type fine | +| `absl::InlinedVector` | requires a complete value type, see the note above | +| `absl::FixedArray` | the size is fixed at construction, so `resize`, `push_back`, `insert` and `erase` are missing | + +## `StringType` + +`StringType` is used **both** for JSON string values and for the keys of JSON objects +(`string_t` and `object_t::key_type`). + +### Always required + +- A member type `value_type` that is one byte wide and `char`-compatible. The library stores and processes UTF-8 + encoded `char` data and hands `data()` to `#!cpp std::strtoull`/`#!cpp std::strtoll`. + `#!cpp std::wstring`, `#!cpp std::u16string`, and `#!cpp std::u32string` are **not** valid choices; see the FAQ on + [wide string handling](../../home/faq.md#wide-string-handling). +- Constructors: default, copy, move, from `#!cpp const char*` (which must not be `#!cpp explicit`), from + `#!cpp (const char*, size_type)`, and from `#!cpp (size_type, char)`; and copy or move assignment. +- Member functions `size()`, `clear()`, `resize(n, c)`, `data()`, `push_back(char)`, and `operator[]` + (const and non-const, returning references). `c_str()` and `back()` are **not** required. +- `data()` must return a pointer to a contiguous, **null-terminated** buffer -- the parser hands it to + `#!cpp std::strtoull`. A type whose `data()` is not null-terminated does not fail to compile; it silently + misparses numbers. +- `append(const char*, size_type)`, used by [`dump`](../../api/basic_json/dump.md), and `append(const StringType&)`, + used by the CBOR reader for indefinite-length strings. The library's internal string concatenation additionally has + to append a `#!cpp char` and a `#!cpp const char*`; for each it selects between `append(arg)`, `#!cpp operator+=`, + `append(first, last)`, and `append(data, size)`. +- The comparison operator `==` against another `StringType`, and `<` for use as a key of the chosen + [`ObjectType`](#objecttype) (with the default comparator, `#!cpp std::less<>` must be able to compare two + `StringType` values, and a `StringType` with the key types used for lookup). `!=` is never applied to a + `StringType`, and `==` against `#!cpp const char*` is resolved by the implicit `#!cpp const char*` constructor. + +### Required for the binary formats + +- `resize(n)`, used by the readers to make room for a block of bytes. +- Non-const `operator[]`, into which the readers `#!cpp std::memcpy` those bytes. A non-`#!cpp const` `data()` would + serve just as well, but `#!cpp std::string` has only had one since C++17, and the library still supports C++11. + +### Required for JSON Pointer, `flatten`, and `diff` + +- A static member `npos` and the member function `find_first_of(char, size_type)` -- together with `data()`, + `reserve(n)`, and `append(const char*, size_type)` they implement the escaping and unescaping of reference tokens + described in RFC 6901. Neither `find(const StringType&, size_type)`, nor `substr(pos, count)`, nor + `replace(pos, count, const StringType&)` is required. +- `empty()`. +- `begin()` and `end()` -- used by + [`operator[](const json_pointer&)`](../../api/basic_json/operator%5B%5D.md) to decide whether a reference token + denotes an array index. + +### Required for other functionality + +| Functionality | Additional requirement | +|-----------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------| +| [`diff`](../../api/basic_json/diff.md), [`items`](../../api/basic_json/items.md), [`std::hash`](../../api/basic_json/std_hash.md) | conversion of a `#!cpp std::size_t` to `StringType`: either assignability from the result of `#!cpp std::to_string`, or an ADL overload `#!cpp void int_to_string(StringType&, std::size_t)` | +| [`std::hash`](../../api/basic_json/std_hash.md) | additionally a specialization of `#!cpp std::hash` | +| [`to_bson`](../../api/basic_json/to_bson.md) | `find(value_type)` and `npos` | +| [`parse`](../../api/basic_json/parse.md) from a `string_t` | the input adapters must accept it; otherwise pass a character range | +| `#!cpp operator<<(std::ostream&, const json_pointer&)` | streamability to `#!cpp std::ostream` | +| exception messages | `data()` and `size()`, or `begin()` and `end()` | + +### Compatible types + +| Type | Notes | +|-----------------------------------------------------------------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------| +| `#!cpp std::string` (default) | | +| `#!cpp std::basic_string` with a custom **stateless** allocator | | +| `#!cpp std::pmr::string` | see the warning below before relying on the memory resource | +| `boost::container::string` | needs a user-supplied `#!cpp std::hash` specialization (Boost provides `boost::hash` instead) | +| `folly::fbstring` | requires C++20, see the note above | +| `eastl::string` | needs a user-supplied `#!cpp std::hash` and an ADL `int_to_string` (it is not assignable from a `#!cpp std::string`); [`parse`](../../api/basic_json/parse.md) does not accept it directly -- pass a character range or a `#!cpp std::string` | +| a custom string class in a user-defined namespace | if the requirements above are met | + +### Types that cannot be used + +| Type | Reason | +|----------------------------------------------------------------------|---------------------------------------------------------------------------------------------------------| +| `#!cpp std::wstring`, `#!cpp std::u16string`, `#!cpp std::u32string` | the character type is not one byte wide | +| `#!cpp std::u8string` | one byte wide, but `#!cpp char8_t` is not `#!cpp char`-compatible | +| `absl::Cord` | no `value_type`, and the storage is not contiguous | +| `QString` | no `append(const char*, size_type)`; its `QChar` is also two bytes wide, though that is never diagnosed | + +!!! warning "A `std::pmr::string` mostly does not use the memory resource you choose" + + `basic_json` cannot be given an allocator or a memory resource. `AllocatorType` is default-constructed at every + allocation and has to be stateless (see [`AllocatorType`](#allocatortype)), and string values the library creates + are constructed with their own default allocator. So: + + - Every string the library itself produces -- from [`parse`](../../api/basic_json/parse.md), from + [`dump`](../../api/basic_json/dump.md), or by default construction -- allocates from + `#!cpp std::pmr::get_default_resource()`. + - **Copying** an arena-backed string into a value silently drops its memory resource: the copy lands on the + default resource, because `#!cpp std::pmr::polymorphic_allocator` does not propagate on copy construction. + Nothing warns about this. + - **Moving** one in does keep it, and later growth still allocates from that arena -- but it does not survive a + copy of the enclosing `basic_json`. + - Passing `#!cpp std::pmr::polymorphic_allocator` as `AllocatorType` does not work around any of this; it does + not compile. + + Apart from moving a string in, the only way to redirect these allocations is the process-global + `#!cpp std::pmr::set_default_resource()`. + +!!! tip "Reference implementation" + + `docs/mkdocs/docs/examples/custom_string_type.hpp` wraps a private `#!cpp std::string` and satisfies every + requirement above -- a good starting point for a custom `StringType`. The unit test + `tests/src/unit-alt-string.cpp` contains a more thorough variant, `alt_string`, exercised against a larger part + of the API. + + ```cpp + --8<-- "examples/custom_string_type.hpp" + ``` + +??? example "Compiling and using it" + + ```cpp + --8<-- "examples/custom_string_type.cpp" + ``` + + Output: + + ```json + --8<-- "examples/custom_string_type.output" + ``` + +## `BooleanType` + +`boolean_t` is stored **directly** inside `basic_json`, as a member of an anonymous union. + +### Always required + +- A literal type that is trivially default-constructible, trivially copyable, and trivially destructible; otherwise the + union's special member functions are deleted. +- **Implicitly** convertible from `#!cpp bool` -- an `#!cpp explicit` constructor is not enough, because the + `to_json` overload for a custom `BooleanType` is constrained on `#!cpp std::is_convertible` -- and contextually + convertible to `#!cpp bool` (here an `#!cpp explicit operator bool` is fine). +- Comparison operators `==`, `!=`, `<`, `<=`, `>`, `>=` (or `<=>`). +- Convertible from and to `#!cpp bool` through the serializer, because + [`get()`](../../api/basic_json/get.md) is used internally. + +There is little reason to use anything other than `#!cpp bool` here. + +### Compatible types + +`#!cpp bool` is the only usable choice. Another trivially copyable type that is implicitly convertible to and from +`#!cpp bool` -- `#!cpp std::uint8_t`, say -- does compile, and JSON booleans still round-trip, but the type then +serves as both `boolean_t` and an ordinary integer: `basic_json` can no longer be constructed or assigned from a +`#!cpp std::uint8_t` at all (the boolean and unsigned-integer `to_json` overloads become ambiguous), and +[`get()`](../../api/basic_json/get.md) on a number throws +[`type_error.302`](../../home/exceptions.md#jsonexceptiontype_error302) instead of returning the value. + +## `NumberIntegerType` and `NumberUnsignedType` + +Both types are stored **directly** inside `basic_json`'s union. + +### Always required + +- `#!cpp std::is_integral` must be satisfied: `NumberIntegerType` must be a **signed** integer type, + `NumberUnsignedType` an **unsigned** integer type. Class types are not supported -- among others, the constructors + taking integer values are constrained on `#!cpp std::is_integral`. +- Trivially default-constructible, trivially copyable, and trivially destructible (union member). +- `#!cpp std::numeric_limits` must be specialized for both types. +- `NumberUnsignedType` must be able to represent the absolute value of every `NumberIntegerType` value; serialization + of negative numbers converts the value to `NumberUnsignedType`. A `#!cpp static_assert` requires it to be at least as + wide as `NumberIntegerType`, which is what that amounts to for the standard integer types. +- Both types must fit into the internal 64-character number buffer used by + [`dump`](../../api/basic_json/dump.md), which is the case for all standard integer types. +- [`std::hash`](../../api/basic_json/std_hash.md) additionally requires `#!cpp std::hash` specializations. + +### Notes + +The number types influence what the parser accepts: an integer literal that does not round-trip through the chosen type +is stored as [`number_float_t`](../../api/basic_json/number_float_t.md) instead. Choosing types narrower than 64 bits +therefore silently changes parse results rather than raising an error. See +[Number Handling](number_handling.md) for details. + +### Compatible types + +| Type pair | Support | +|----------------------------------------------------------------------------------------------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------| +| `#!cpp std::int64_t` / `#!cpp std::uint64_t` (default) | full | +| `#!cpp std::int32_t` / `#!cpp std::uint32_t`, `#!cpp long long` / `#!cpp unsigned long long` | full; narrower types change which literals the parser can represent | +| any other pair of standard signed/unsigned integer types | full | +| class types, enumerations | not usable; `#!cpp std::is_integral` must hold | +| `#!cpp bool`, or a type already used for another member of the union | not usable; `#!cpp std::is_integral` is in fact `#!cpp true`, but the `get_impl_ptr` overloads for `boolean_t`, `number_integer_t`, `number_unsigned_t` and `number_float_t` would collide | + +## `NumberFloatType` + +`number_float_t` is stored **directly** inside `basic_json`'s union. + +### Always required + +- Trivially default-constructible, trivially copyable, and trivially destructible (union member). +- `#!cpp std::numeric_limits` must be specialized; `max_digits10` is used to size the conversion. +- `#!cpp std::isfinite` must be applicable to the type. + +### Required for parsing and serialization + +`NumberFloatType` must be one of `#!cpp float`, `#!cpp double`, or `#!cpp long double`: + +- The [parser](../parsing/index.md) converts number literals with `#!cpp std::strtof`, `#!cpp std::strtod`, or + `#!cpp std::strtold`; the library provides overloads for exactly these three types. +- [`dump`](../../api/basic_json/dump.md) falls back to `#!cpp std::snprintf` with the `%g` and `%Lg` conversion + specifiers, for which the library likewise provides only `#!cpp double` and `#!cpp long double` overloads + (`#!cpp float` is promoted to `#!cpp double`). + +If `#!cpp std::numeric_limits` describes an IEEE 754 binary32 or binary64 number, `dump` uses the +Grisu2 algorithm, which produces the shortest representation that round-trips. Otherwise the `snprintf` fallback with +`max_digits10` digits is used. + +### Required for the binary formats + +`NumberFloatType` must be `#!cpp float` or `#!cpp double`. The writers for +[CBOR, MessagePack, UBJSON, BJData, and BSON](../binary_formats/index.md) map a floating-point value onto an IEEE 754 +binary32 or binary64 field and have no encoding for `#!cpp long double`. + +### Compatible types + +| Type | Support | +|--------------------------|-----------------------------------------------------------------------------------------------------------------------| +| `#!cpp double` (default) | full; short round-trip output through Grisu2 | +| `#!cpp float` | full; short round-trip output through Grisu2 | +| `#!cpp long double` | `dump` and `parse` only; the binary format writers do not compile, as they only handle IEEE 754 binary32 and binary64 | +| any other type | not usable | + +## `AllocatorType` + +`AllocatorType` is instantiated with **one** argument, for each of `object_t`, `array_t`, `string_t`, `binary_t`, +`basic_json`, and `#!cpp std::pair`. + +### Always required + +- The template must be usable with exactly one type argument. The library instantiates `AllocatorType` directly and + never uses `#!cpp std::allocator_traits<...>::rebind_alloc`. +- It must satisfy the [Allocator](https://en.cppreference.com/w/cpp/named_req/Allocator) named requirement so that + `#!cpp std::allocator_traits` can be used with it. +- It must be **default-constructible and stateless**. Objects are allocated with a default-constructed allocator and + deallocated with a *different* default-constructed allocator, and + [`get_allocator()`](../../api/basic_json/get_allocator.md) returns a default-constructed instance. Allocators + carrying state are not supported, so there is no way to tell a `basic_json` where to allocate from; see the note + under [`StringType`](#stringtype) for what that means in practice. A stateful allocator is **not diagnosed**: it + compiles and silently ignores the state. +- It must support **incomplete types**: `AllocatorType` is instantiated inside the definition of + `basic_json` itself. +- `#!cpp std::allocator_traits>::pointer` becomes + [`basic_json::pointer`](../../api/basic_json/index.md#container-types), and iterators are constructed from raw + `#!cpp basic_json*` values. The `pointer` type must therefore be a plain pointer; fancy pointers are not supported. + +### Compatible types + +| Type | Support | +|-------------------------------------------------------------------|----------------------------------------| +| `#!cpp std::allocator` (default) | full | +| a custom stateless allocator template | full | +| stateful allocators, e.g. `#!cpp std::pmr::polymorphic_allocator` | not usable; see the requirements above | + +## `JSONSerializer` + +`JSONSerializer` is instantiated as `JSONSerializer` and defaults to +[`adl_serializer`](../../api/adl_serializer/index.md). + +### Always required + +- The template must accept **two** type arguments. It does not have to give the second one a default -- `basic_json` + declares the parameter as `#!cpp template class JSONSerializer`, so uses such as + `#!cpp JSONSerializer` inside the library supply `#!cpp void` themselves. The second parameter exists so that + partial specializations can be constrained by SFINAE. +- For every type `T` that is converted **to** a JSON value, a static member function + `#!cpp static void to_json(basic_json&, T)` must exist. +- For every type `T` that is converted **from** a JSON value, either + `#!cpp static void from_json(const basic_json&, T&)` or `#!cpp static T from_json(const basic_json&)` must exist. + The latter form is required for types that are not default-constructible; see + [Arbitrary Types Conversions](../arbitrary_types.md). +- To support the [converting constructor](../../api/basic_json/basic_json.md) between different `basic_json` + specializations, `to_json` must be available for `boolean_t`, `number_integer_t`, `number_unsigned_t`, + `number_float_t`, `string_t`, `object_t`, `array_t`, and `binary_t` of the *source* specialization. + +### Compatible types + +| Type | Support | +|---------------------------------------------------------------------------|-------------------------------------------------------------------| +| [`nlohmann::adl_serializer`](../../api/adl_serializer/index.md) (default) | full | +| a class template deriving from `adl_serializer` | full; the usual way to change behavior while keeping the defaults | +| an unrelated template with the same interface | full, but it has to handle every type the library converts | + +## `BinaryType` + +`BinaryType` is not a JSON type; it is used for the byte strings of the +[binary formats](../binary_formats/index.md). It is wrapped as + +```cpp +using binary_t = nlohmann::byte_container_with_subtype; +``` + +### Always required + +- A non-`final` class type -- [`byte_container_with_subtype`](../../api/byte_container_with_subtype/index.md) derives + from it publicly. +- A member type `value_type` that is **exactly one byte** wide (e.g., `#!cpp std::uint8_t`, `#!cpp char`, or + `#!cpp std::byte`). Readers and writers reinterpret the container's storage as raw bytes, so a wider `value_type` is + rejected with a `#!cpp static_assert`. +- Contiguous storage: the binary readers `#!cpp std::memcpy` into `#!cpp &binary[n]`, the writers `reinterpret_cast` + `data()`. `#!cpp data() + n` would do for the readers too, but they share one helper with + [`StringType`](#stringtype), whose non-`#!cpp const` `data()` is C++17 and later only. +- Default-constructible, copy-constructible, and move-constructible. +- Member functions `size()`, `empty()`, `data()`, `resize()`, `operator[]`, `back()`, `begin()`, `end()`, `cbegin()`, + and `cend()` with random-access iterators, and `insert(pos, first, last)`, which the CBOR reader uses to join the + chunks of an indefinite-length byte string. `push_back()` is **not** required. +- Comparison operators: `==` is used by + [`byte_container_with_subtype`](../../api/byte_container_with_subtype/index.md), the relational operators by + [`basic_json`'s comparison operators](../../api/basic_json/operator_le.md). + +### Required for individual functions + +- `clear()`, for [`basic_json::clear()`](../../api/basic_json/clear.md). + +`max_size()`, `at()`, `reserve()`, `erase()`, `pop_back()`, and `emplace_back()` are **not** used at all. + +See [`binary_t`](../../api/basic_json/binary_t.md) for how a non-default `BinaryType` changes the meaning of assigning +such a container to a `basic_json` value. + +!!! tip "Reference implementation" + + `docs/mkdocs/docs/examples/custom_binary_type.hpp` wraps a private `#!cpp std::vector` and satisfies + every requirement above -- a good starting point for a custom `BinaryType`. + + ```cpp + --8<-- "examples/custom_binary_type.hpp" + ``` + +??? example "Compiling and using it" + + ```cpp + --8<-- "examples/custom_binary_type.cpp" + ``` + + Output: + + ```json + --8<-- "examples/custom_binary_type.output" + ``` + +### Compatible containers + +| Container | Notes | +|---------------------------------------------------------------------------------------------|---------------------------------------------------------------------------| +| `#!cpp std::vector` (default) | | +| `#!cpp std::vector`, `#!cpp std::vector` | `dump()` writes the bytes as 0..255 whichever is used | +| `boost::container::vector`, `boost::container::small_vector` | | +| `absl::InlinedVector` | usable here, unlike as an `ArrayType`, because the value type is complete | +| `eastl::vector` | usable here, unlike as an `ArrayType`, because `max_size()` is not needed | +| `folly::fbvector` | requires C++20, see the note above | + +### Containers that cannot be used + +| Container | Reason | +|------------------------------------------------------|----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------| +| `QByteArray` | no `empty()` (it spells that `isEmpty()`); its `insert` takes an index rather than an iterator; and it converts to `string_t`, which makes `to_json` ambiguous between a string and a binary value | +| `#!cpp std::string` | `binary_t::container_type` and `string_t` would be the same type, so the two [`swap`](../../api/basic_json/swap.md) overloads collide and `basic_json` cannot be instantiated at all | +| `#!cpp std::deque` | storage is not contiguous, so there is no `data()` | +| containers whose `value_type` is wider than one byte | see above -- accepted by the compiler, wrong at runtime | + +## `CustomBaseClass` + +`CustomBaseClass` is an extension point: unless it is `#!cpp void` (the default, which selects the empty +`nlohmann::json_default_base`), `basic_json` publicly derives from it. + +### Always required + +- A non-`final`, default-constructible class type. +- `basic_json` is copy-/move-constructible and copy-/move-assignable only if `CustomBaseClass` is. + +### Notes + +`basic_json` is documented to be a +[StandardLayoutType](https://en.cppreference.com/w/cpp/named_req/StandardLayoutType). Because `basic_json` has +non-static data members of its own, a `CustomBaseClass` with non-static data members forfeits this guarantee. + +Note the namespace of `CustomBaseClass` becomes an associated namespace of `basic_json` for the purpose of +argument-dependent lookup. + +See [`json_base_class_t`](../../api/basic_json/json_base_class_t.md) for an example. + +### Compatible types + +| Type | Support | +|----------------------------------------------|----------------------------------------------------------------------------| +| `#!cpp void` (default) | an empty base class is used; no effect on `basic_json` | +| any default-constructible, non-`final` class | full; see [`json_base_class_t`](../../api/basic_json/json_base_class_t.md) | + +## Cross-specialization conversions + +Converting a value from one `basic_json` specialization into another (see the +[converting constructor](../../api/basic_json/basic_json.md)) imposes two additional requirements that are not +diagnosed at compile time. With assertions enabled they abort on the `#!cpp JSON_ASSERT` at the end of the converting +constructor; under `#!cpp NDEBUG` they fail **silently** at runtime: + +- The target `string_t` must be directly constructible from the source `string_t`. Otherwise the string is converted to + an array of character codes. +- The target `object_t::key_type` must be directly constructible from the source object's key type. Otherwise the + object is converted to an array of key/value pairs. + +See [issue #3425](https://github.com/nlohmann/json/issues/3425), [`string_t`](../../api/basic_json/string_t.md), and +[`object_t`](../../api/basic_json/object_t.md). + +## See also + +- [Types](index.md) -- overview of how JSON values are stored +- [Number Handling](number_handling.md) -- how the number types affect parsing and serialization +- [Object Order](../object_order.md) -- using an insertion-ordered `ObjectType` +- [`basic_json`](../../api/basic_json/index.md) -- API documentation of the class template diff --git a/docs/mkdocs/mkdocs.yml b/docs/mkdocs/mkdocs.yml index 5c9f72fc3..d0f9cfdfd 100644 --- a/docs/mkdocs/mkdocs.yml +++ b/docs/mkdocs/mkdocs.yml @@ -98,6 +98,7 @@ nav: - Types: - features/types/index.md - features/types/number_handling.md + - features/types/template_parameters.md - Integration: - integration/index.md - integration/migration_guide.md diff --git a/include/nlohmann/detail/exceptions.hpp b/include/nlohmann/detail/exceptions.hpp index 3e5b45101..808a3b6bf 100644 --- a/include/nlohmann/detail/exceptions.hpp +++ b/include/nlohmann/detail/exceptions.hpp @@ -101,7 +101,10 @@ class exception : public std::exception { if (&element.second == current) { - tokens.emplace_back(element.first.c_str()); + // data() is null-terminated, so a key containing + // a null byte is cut short here rather than + // truncating the whole message at what() + tokens.emplace_back(element.first.data()); break; } } diff --git a/include/nlohmann/detail/hash.hpp b/include/nlohmann/detail/hash.hpp index 61b3469f1..be8063f89 100644 --- a/include/nlohmann/detail/hash.hpp +++ b/include/nlohmann/detail/hash.hpp @@ -114,7 +114,9 @@ std::size_t hash(const BasicJsonType& j) seed = combine(seed, static_cast(j.get_binary().subtype())); for (const auto byte : j.get_binary()) { - seed = combine(seed, std::hash {}(byte)); + // the cast is needed for binary types whose value type is not + // an integer (e.g., std::byte) + seed = combine(seed, std::hash {}(static_cast(byte))); } return seed; } diff --git a/include/nlohmann/detail/input/binary_reader.hpp b/include/nlohmann/detail/input/binary_reader.hpp index e0343fd0b..df46eea58 100644 --- a/include/nlohmann/detail/input/binary_reader.hpp +++ b/include/nlohmann/detail/input/binary_reader.hpp @@ -3146,7 +3146,10 @@ class binary_reader number_string, out_of_range::create(406, concat("number overflow parsing '", number_string, '\''), nullptr)); } - return sax->number_float(parsed_float, std::move(number_string)); + // number_string is a std::string, while the SAX interface takes a + // string_t; convert explicitly, as the two are only implicitly + // convertible for some string types + return sax->number_float(parsed_float, string_t(number_string.data(), number_string.size())); } case token_type::uninitialized: case token_type::literal_true: diff --git a/include/nlohmann/detail/iterators/iter_impl.hpp b/include/nlohmann/detail/iterators/iter_impl.hpp index 44448611a..22f3ffc39 100644 --- a/include/nlohmann/detail/iterators/iter_impl.hpp +++ b/include/nlohmann/detail/iterators/iter_impl.hpp @@ -88,8 +88,13 @@ class iter_impl // NOLINT(cppcoreguidelines-special-member-functions,hicpp-speci iter_impl() = default; ~iter_impl() = default; - iter_impl(iter_impl&&) noexcept = default; - iter_impl& operator=(iter_impl&&) noexcept = default; + // the exception specification is left to be computed rather than declared: + // an array or object type whose iterator is not nothrow move constructible + // (std::deque's is not before libstdc++ 11) would make a declared noexcept + // differ from the implicit one, which deletes the function -- and is an + // error outright with older compilers + iter_impl(iter_impl&&) = default; // NOLINT(hicpp-noexcept-move,performance-noexcept-move-constructor,cppcoreguidelines-noexcept-move-operations) + iter_impl& operator=(iter_impl&&) = default; // NOLINT(hicpp-noexcept-move,performance-noexcept-move-constructor,cppcoreguidelines-noexcept-move-operations) /*! @brief constructor for a given JSON instance diff --git a/include/nlohmann/detail/json_pointer.hpp b/include/nlohmann/detail/json_pointer.hpp index 247c5babb..1540a8d6f 100644 --- a/include/nlohmann/detail/json_pointer.hpp +++ b/include/nlohmann/detail/json_pointer.hpp @@ -17,6 +17,7 @@ #endif // JSON_NO_IO #include // max #include // accumulate +#include // set #include // string #include // move #include // vector @@ -71,7 +72,7 @@ class json_pointer string_t{}, [](const string_t& a, const string_t& b) { - return detail::concat(a, '/', detail::escape(b)); + return detail::concat(a, '/', detail::escape(b)); }); } @@ -265,7 +266,7 @@ class json_pointer JSON_THROW(detail::parse_error::create(109, 0, detail::concat("array index '", s, "' is not a number"), nullptr)); } - const char* p = s.c_str(); + const char* p = s.data(); char* p_end = nullptr; // NOLINT(misc-const-correctness) errno = 0; // strtoull doesn't reset errno const unsigned long long res = std::strtoull(p, &p_end, 10); // NOLINT(runtime/int) @@ -300,19 +301,35 @@ class json_pointer } private: + /*! + @brief the reference token sequences that denote arrays + + @ref unflatten collects the pointer prefixes that have a reference token 0 + among their children; @ref get_and_create creates arrays exactly below + those prefixes and objects everywhere else. Deciding this up front keeps + the result independent of the order in which the flattened object is + iterated, which is unspecified for some object types. + */ + using array_parents_t = std::set>; + /*! @brief create and return a reference to the pointed to value @complexity Linear in the number of reference tokens. + @throw parse_error.106 if an array index begins with '0' @throw parse_error.109 if array index is not a number @throw type_error.313 if value cannot be unflattened */ template - BasicJsonType& get_and_create(BasicJsonType& j) const + BasicJsonType& get_and_create(BasicJsonType& j, const array_parents_t& array_parents) const { auto* result = &j; + // the reference tokens that have been consumed so far; used to look up + // whether the value to be created below is an array or an object + std::vector prefix; + // in case no reference tokens exist, return a reference to the JSON value // j which will be overwritten by a primitive value for (const auto& reference_token : reference_tokens) @@ -321,10 +338,11 @@ class json_pointer { case detail::value_t::null: { - if (reference_token == "0") + if (array_parents.find(prefix) != array_parents.end()) { - // start a new array if the reference token is 0 - result = &result->operator[](0); + // some reference token below this position is 0, so the + // value is an array + result = &result->operator[](array_index(reference_token)); } else { @@ -364,6 +382,8 @@ class json_pointer default: JSON_THROW(detail::type_error::create(313, "invalid value to unflatten", &j)); } + + prefix.push_back(reference_token); } return *result; @@ -837,7 +857,8 @@ class json_pointer { // use the text between the beginning of the reference token // (start) and the last slash (slash). - auto reference_token = reference_string.substr(start, slash - start); + const auto count = (slash == string_t::npos ? reference_string.size() : slash) - start; + auto reference_token = string_t(reference_string.data() + start, count); // check reference tokens are properly escaped for (std::size_t pos = reference_token.find_first_of('~'); @@ -953,6 +974,24 @@ class json_pointer BasicJsonType result; + // collect the pointer prefixes that have a reference token 0 among + // their children; the values below them are arrays, all others are + // objects (see array_parents_t) + array_parents_t array_parents; + for (const auto& element : *value.m_data.m_value.object) + { + json_pointer ptr(element.first); + std::vector prefix; + for (auto& reference_token : ptr.reference_tokens) + { + if (reference_token == "0") + { + array_parents.insert(prefix); + } + prefix.push_back(std::move(reference_token)); + } + } + // iterate the JSON object values for (const auto& element : *value.m_data.m_value.object) { @@ -965,7 +1004,7 @@ class json_pointer // that if the JSON pointer is "" (i.e., points to the whole value), // function get_and_create returns a reference to the result itself. // An assignment will then create a primitive value. - json_pointer(element.first).get_and_create(result) = element.second; + json_pointer(element.first).get_and_create(result, array_parents) = element.second; } return result; diff --git a/include/nlohmann/detail/meta/type_traits.hpp b/include/nlohmann/detail/meta/type_traits.hpp index ebf6a2c26..6f8bf2a3d 100644 --- a/include/nlohmann/detail/meta/type_traits.hpp +++ b/include/nlohmann/detail/meta/type_traits.hpp @@ -172,17 +172,18 @@ struct has_to_json < BasicJsonType, T, enable_if_t < !is_basic_json::value >> template using detect_key_compare = typename T::key_compare; -template -struct has_key_compare : std::integral_constant::value> {}; - -// obtains the actual object key comparator +// obtains the actual object key comparator: object_t::key_compare if the +// object type defines it, and default_object_comparator_t otherwise +// +// note detected_or_t is used rather than std::conditional, because the latter +// names both of its type arguments eagerly; object_t::key_compare would then +// be a hard error for an object type that does not define it template struct actual_object_comparator { using object_t = typename BasicJsonType::object_t; using object_comparator_t = typename BasicJsonType::default_object_comparator_t; - using type = typename std::conditional < has_key_compare::value, - typename object_t::key_compare, object_comparator_t>::type; + using type = detected_or_t; }; template @@ -778,6 +779,22 @@ using has_erase_with_key_type = typename std::conditional < std::true_type, std::false_type >::type; +template +using detect_erase_with_iterator = decltype(std::declval().erase(std::declval())); + +// type trait to check if erase(iterator) returns void instead of the following +// iterator, as the object types that do not compute a successor the caller may +// not need do +template +using erase_returns_void = is_detected_exact; + +template +using detect_capacity = decltype(std::declval().capacity()); + +// type trait to check if a type has a capacity() member function +template +struct has_capacity : std::integral_constant::value> {}; + // a naive helper to check if a type is an ordered_map (exploits the fact that // ordered_map inherits capacity() from std::vector) template diff --git a/include/nlohmann/detail/output/binary_writer.hpp b/include/nlohmann/detail/output/binary_writer.hpp index 28290de3e..e9ccd23b5 100644 --- a/include/nlohmann/detail/output/binary_writer.hpp +++ b/include/nlohmann/detail/output/binary_writer.hpp @@ -261,7 +261,7 @@ class binary_writer // step 2: write the string oa->write_characters( - reinterpret_cast(j.m_data.m_value.string->c_str()), + reinterpret_cast(j.m_data.m_value.string->data()), j.m_data.m_value.string->size()); break; } @@ -581,7 +581,7 @@ class binary_writer // step 2: write the string oa->write_characters( - reinterpret_cast(j.m_data.m_value.string->c_str()), + reinterpret_cast(j.m_data.m_value.string->data()), j.m_data.m_value.string->size()); break; } @@ -798,7 +798,7 @@ class binary_writer } write_number_with_ubjson_prefix(j.m_data.m_value.string->size(), true, use_bjdata); oa->write_characters( - reinterpret_cast(j.m_data.m_value.string->c_str()), + reinterpret_cast(j.m_data.m_value.string->data()), j.m_data.m_value.string->size()); break; } @@ -897,7 +897,9 @@ class binary_writer for (size_t i = 0; i < j.m_data.m_value.binary->size(); ++i) { oa->write_character(to_char_type(bjdata_draft3 ? 'B' : 'U')); - oa->write_character(to_char_type(j.m_data.m_value.binary->data()[i])); + // the cast is needed for binary types whose value type + // is not an integer (e.g., std::byte) + oa->write_character(to_char_type(static_cast(j.m_data.m_value.binary->data()[i]))); } } @@ -958,7 +960,7 @@ class binary_writer { write_number_with_ubjson_prefix(el.first.size(), true, use_bjdata); oa->write_characters( - reinterpret_cast(el.first.c_str()), + reinterpret_cast(el.first.data()), el.first.size()); write_ubjson(el.second, use_count, use_type, prefix_required, use_bjdata, bjdata_version); } @@ -1021,8 +1023,11 @@ class binary_writer { oa->write_character(to_char_type(element_type)); oa->write_characters( - reinterpret_cast(name.c_str()), - name.size() + 1u); + reinterpret_cast(name.data()), + name.size()); + // the terminating null byte is written explicitly rather than taken + // from the buffer, so that string_t::data() need not be null-terminated + oa->write_character(to_char_type(0x00)); } /*! @@ -1063,8 +1068,11 @@ class binary_writer write_number(to_bson_length(value.size() + 1ul), true); oa->write_characters( - reinterpret_cast(value.c_str()), - value.size() + 1); + reinterpret_cast(value.data()), + value.size()); + // the terminating null byte is written explicitly rather than taken + // from the buffer, so that string_t::data() need not be null-terminated + oa->write_character(to_char_type(0x00)); } /*! @@ -1155,7 +1163,11 @@ class binary_writer const std::size_t embedded_document_size = std::accumulate(std::begin(value), std::end(value), static_cast(0), [&array_index](std::size_t result, const typename BasicJsonType::array_t::value_type & el) { - return result + calc_bson_element_size(std::to_string(array_index++), el); + // the index is built as a std::string, while calc_bson_element_size + // takes a string_t; convert explicitly, as the two are only + // implicitly convertible for some string types + const auto key = std::to_string(array_index++); + return result + calc_bson_element_size(string_t(key.data(), key.size()), el); }); return sizeof(std::int32_t) + embedded_document_size + 1ul; @@ -1182,7 +1194,11 @@ class binary_writer for (const auto& el : value) { - write_bson_element(std::to_string(array_index++), el); + // the index is built as a std::string, while write_bson_element takes + // a string_t; convert explicitly, as the two are only implicitly + // convertible for some string types + const auto key = std::to_string(array_index++); + write_bson_element(string_t(key.data(), key.size()), el); } oa->write_character(to_char_type(0x00)); diff --git a/include/nlohmann/detail/output/serializer.hpp b/include/nlohmann/detail/output/serializer.hpp index 9560729ad..f9e7f7840 100644 --- a/include/nlohmann/detail/output/serializer.hpp +++ b/include/nlohmann/detail/output/serializer.hpp @@ -1061,7 +1061,7 @@ class serializer { case error_handler_t::strict: { - JSON_THROW(type_error::create(316, concat("incomplete UTF-8 string; last byte: 0x", hex_bytes(static_cast(s.back() | 0))), nullptr)); + JSON_THROW(type_error::create(316, concat("incomplete UTF-8 string; last byte: 0x", hex_bytes(static_cast(s[s.size() - 1] | 0))), nullptr)); } case error_handler_t::ignore: @@ -1322,6 +1322,19 @@ class serializer pos += 6; } + /*! + @brief convert a single element of a binary value to its byte value + + The elements of a binary value are dumped as the numbers 0..255, regardless + of the value type of the configured BinaryType: that type may be signed + (`char`), unsigned (`std::uint8_t`), or not an integer at all + (`std::byte`), none of which @ref dump_integer can handle uniformly. + */ + static std::uint8_t to_byte_value(binary_char_t x) noexcept + { + return static_cast(x); + } + // templates to avoid warnings about useless casts template ::value, int> = 0> bool is_negative_number(NumberType x) @@ -1343,8 +1356,10 @@ class serializer an arbitrary number, and the three digits it takes at most are written straight into the write buffer. - Any byte type that is not a plain unsigned byte is left to @ref dump_integer, - whose representation of it may differ. + Any byte type that is not a plain unsigned byte is converted to its + @ref to_byte_value "byte value" and left to @ref dump_integer, so a signed + or non-integral BinaryType::value_type (`char`, `std::byte`, ...) still + dumps as 0..255. */ template void dump_byte(const ByteType value) @@ -1357,7 +1372,7 @@ class serializer template void dump_byte(const ByteType value, std::false_type /*is_plain_byte*/) { - dump_integer(value); + dump_integer(to_byte_value(value)); } template @@ -1403,8 +1418,7 @@ class serializer template < typename NumberType, detail::enable_if_t < std::is_integral::value || std::is_same::value || - std::is_same::value || - std::is_same::value, + std::is_same::value, int > = 0 > void dump_integer(NumberType x) { diff --git a/include/nlohmann/detail/string_escape.hpp b/include/nlohmann/detail/string_escape.hpp index 7715dde7a..0d24a56bc 100644 --- a/include/nlohmann/detail/string_escape.hpp +++ b/include/nlohmann/detail/string_escape.hpp @@ -8,50 +8,56 @@ #pragma once +#include // size_t + #include NLOHMANN_JSON_NAMESPACE_BEGIN namespace detail { -/*! -@brief replace all occurrences of a substring by another string - -@param[in,out] s the string to manipulate; changed so that all - occurrences of @a f are replaced with @a t -@param[in] f the substring to replace with @a t -@param[in] t the string to replace @a f - -@pre The search string @a f must not be empty. **This precondition is -enforced with an assertion.** - -@since version 2.0.0 -*/ -template -inline void replace_substring(StringType& s, const StringType& f, - const StringType& t) -{ - JSON_ASSERT(!f.empty()); - for (auto pos = s.find(f); // find the first occurrence of f - pos != StringType::npos; // make sure f was found - s.replace(pos, f.size(), t), // replace with t, and - pos = s.find(f, pos + t.size())) // find the next occurrence of f - {} -} - /*! * @brief string escaping as described in RFC 6901 (Sect. 4) * @param[in] s string to escape * @return escaped string * * Note the order of escaping "~" to "~0" and "/" to "~1" is important. + * + * The string is rebuilt in a single pass, appending whole runs between the + * characters that need escaping. Scanning with find_first_of() keeps the + * common case -- nothing to escape -- as fast as a single search, while + * repeated replace() calls would move the tail of the string once per + * escaped character. */ template -inline StringType escape(StringType s) +inline StringType escape(const StringType& s) { - replace_substring(s, StringType{"~"}, StringType{"~0"}); - replace_substring(s, StringType{"/"}, StringType{"~1"}); - return s; + auto next_special = [&s](std::size_t from) + { + const auto tilde = s.find_first_of('~', from); + const auto slash = s.find_first_of('/', from); + return tilde < slash ? tilde : slash; // npos is the largest value + }; + + auto pos = next_special(0); + if (pos == StringType::npos) + { + return s; + } + + StringType result; + result.reserve(s.size() + 2); + + std::size_t run = 0; + while (pos != StringType::npos) + { + result.append(s.data() + run, pos - run); + result.append(s[pos] == '~' ? "~0" : "~1", 2); + run = pos + 1; + pos = next_special(run); + } + result.append(s.data() + run, s.size() - run); + return result; } /*! @@ -60,12 +66,43 @@ inline StringType escape(StringType s) * @return unescaped string * * Note the order of escaping "~1" to "/" and "~0" to "~" is important. + * + * Rebuilt in a single pass, see @ref escape. A "~" that is followed by + * neither "0" nor "1" is passed through unchanged; @ref json_pointer rejects + * such input before it gets here. */ template inline void unescape(StringType& s) { - replace_substring(s, StringType{"~1"}, StringType{"/"}); - replace_substring(s, StringType{"~0"}, StringType{"~"}); + auto pos = s.find_first_of('~', 0); + if (pos == StringType::npos) + { + return; + } + + StringType result; + result.reserve(s.size()); + + std::size_t run = 0; + while (pos != StringType::npos) + { + result.append(s.data() + run, pos - run); + + const auto next = pos + 1; + if (next < s.size() && (s[next] == '0' || s[next] == '1')) + { + result.append(s[next] == '0' ? "~" : "/", 1); + run = pos + 2; + } + else + { + result.append("~", 1); + run = pos + 1; + } + pos = s.find_first_of('~', run); + } + result.append(s.data() + run, s.size() - run); + s = result; } } // namespace detail diff --git a/include/nlohmann/json.hpp b/include/nlohmann/json.hpp index ca3cda17a..1aafbf78a 100644 --- a/include/nlohmann/json.hpp +++ b/include/nlohmann/json.hpp @@ -404,6 +404,18 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @} + // Two template parameter requirements that would otherwise be silently + // violated: neither produces a diagnostic of its own, and both corrupt + // values rather than failing. + + static_assert(sizeof(typename BinaryType::value_type) == 1, + "BinaryType::value_type must be exactly one byte wide, " + "because the binary readers and writers reinterpret the container's storage as raw bytes"); + + static_assert(sizeof(NumberUnsignedType) >= sizeof(NumberIntegerType), + "NumberUnsignedType must be at least as wide as NumberIntegerType, " + "because it has to hold the absolute value of every NumberIntegerType value"); + private: /// helper for exception-safe object creation @@ -784,21 +796,76 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec return it; } - reference set_parent(reference j, std::size_t old_capacity = detail::unknown_size()) + /// @brief erase an element from the object and return the following one + /// Not every map returns an iterator from erase(iterator): some containers + /// (e.g., Abseil's hash maps) return void to avoid computing a successor + /// the caller may not need. Compute it before erasing for those. + template < typename It, detail::enable_if_t < + !detail::erase_returns_void::value, int > = 0 > + typename object_t::iterator erase_from_object(It pos) + { + return m_data.m_value.object->erase(pos); + } + + template < typename It, detail::enable_if_t < + detail::erase_returns_void::value, int > = 0 > + typename object_t::iterator erase_from_object(It pos) + { + auto next = std::next(pos); + m_data.m_value.object->erase(pos); + return next; + } + + /// @brief the capacity of the stored array, or unknown_size() + /// Only JSON_DIAGNOSTICS uses the value, to detect a reallocation that + /// would invalidate the parent pointers. Array types that do not have a + /// capacity() member function report unknown_size(), which is treated as + /// "the elements may have moved". +#if JSON_DIAGNOSTICS + template < typename A = array_t, detail::enable_if_t < detail::has_capacity::value, int > = 0 > + std::size_t array_capacity() const noexcept + { + return m_data.m_value.array->capacity(); + } + + template < typename A = array_t, detail::enable_if_t < !detail::has_capacity::value, int > = 0 > + std::size_t array_capacity() const noexcept + { + return detail::unknown_size(); + } +#else + static constexpr std::size_t array_capacity() noexcept + { + return detail::unknown_size(); + } +#endif + + /// @brief set the parent of a value that has just been added to an array + /// @param j the added value + /// @param old_capacity the value @ref array_capacity() returned before the + /// insertion + reference set_parent_after_array_insert(reference j, std::size_t old_capacity) { #if JSON_DIAGNOSTICS - if (old_capacity != detail::unknown_size()) + // see https://github.com/nlohmann/json/issues/2838 + JSON_ASSERT(type() == value_t::array); + if (JSON_HEDLEY_UNLIKELY(old_capacity == detail::unknown_size() + || array_capacity() != old_capacity)) { - // see https://github.com/nlohmann/json/issues/2838 - JSON_ASSERT(type() == value_t::array); - if (JSON_HEDLEY_UNLIKELY(m_data.m_value.array->capacity() != old_capacity)) - { - // capacity has changed: update all parents - set_parents(); - return j; - } + // the capacity has changed, or the array type does not let us tell: + // the elements may have moved, so update all parents + set_parents(); + return j; } +#else + static_cast(old_capacity); +#endif + return set_parent(j); + } + reference set_parent(reference j) + { +#if JSON_DIAGNOSTICS // ordered_json uses a vector internally, so pointers could have // been invalidated; see https://github.com/nlohmann/json/issues/2962 #ifdef JSON_HEDLEY_MSVC_VERSION @@ -817,7 +884,6 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec j.m_parent = this; #else static_cast(j); - static_cast(old_capacity); #endif return j; } @@ -2029,22 +2095,17 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec reference at(size_type idx) { // at only works for arrays - if (JSON_HEDLEY_LIKELY(is_array())) - { - JSON_TRY - { - return set_parent(m_data.m_value.array->at(idx)); - } - JSON_CATCH (std::out_of_range&) - { - // create a better exception explanation - JSON_THROW(out_of_range::create(401, detail::concat("array index ", std::to_string(idx), " is out of range"), this)); - } // cppcheck-suppress[missingReturn] - } - else + if (JSON_HEDLEY_UNLIKELY(!is_array())) { JSON_THROW(type_error::create(304, detail::concat("cannot use at() with ", type_name()), this)); } + + if (JSON_HEDLEY_UNLIKELY(idx >= m_data.m_value.array->size())) + { + JSON_THROW(out_of_range::create(401, detail::concat("array index ", std::to_string(idx), " is out of range"), this)); + } + + return set_parent((*m_data.m_value.array)[idx]); } /// @brief access specified array element with bounds checking @@ -2052,22 +2113,17 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const_reference at(size_type idx) const { // at only works for arrays - if (JSON_HEDLEY_LIKELY(is_array())) - { - JSON_TRY - { - return m_data.m_value.array->at(idx); - } - JSON_CATCH (std::out_of_range&) - { - // create a better exception explanation - JSON_THROW(out_of_range::create(401, detail::concat("array index ", std::to_string(idx), " is out of range"), this)); - } // cppcheck-suppress[missingReturn] - } - else + if (JSON_HEDLEY_UNLIKELY(!is_array())) { JSON_THROW(type_error::create(304, detail::concat("cannot use at() with ", type_name()), this)); } + + if (JSON_HEDLEY_UNLIKELY(idx >= m_data.m_value.array->size())) + { + JSON_THROW(out_of_range::create(401, detail::concat("array index ", std::to_string(idx), " is out of range"), this)); + } + + return (*m_data.m_value.array)[idx]; } /// @brief access specified object element with bounds checking @@ -2167,12 +2223,13 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec #if JSON_DIAGNOSTICS // remember array size & capacity before resizing const auto old_size = m_data.m_value.array->size(); - const auto old_capacity = m_data.m_value.array->capacity(); + const auto old_capacity = array_capacity(); #endif m_data.m_value.array->resize(idx + 1); #if JSON_DIAGNOSTICS - if (JSON_HEDLEY_UNLIKELY(m_data.m_value.array->capacity() != old_capacity)) + if (JSON_HEDLEY_UNLIKELY(old_capacity == detail::unknown_size() + || array_capacity() != old_capacity)) { // capacity has changed: update all parents set_parents(); @@ -2563,7 +2620,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec case value_t::object: { - result.m_it.object_iterator = m_data.m_value.object->erase(pos.m_it.object_iterator); + result.m_it.object_iterator = erase_from_object(pos.m_it.object_iterator); break; } @@ -3202,9 +3259,9 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } // add the element to the array (move semantics) - const auto old_capacity = m_data.m_value.array->capacity(); + const auto old_capacity = array_capacity(); m_data.m_value.array->push_back(std::move(val)); - set_parent(m_data.m_value.array->back(), old_capacity); + set_parent_after_array_insert(m_data.m_value.array->back(), old_capacity); // if val is moved from, basic_json move constructor marks it null, so we do not call the destructor } @@ -3235,9 +3292,9 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } // add the element to the array - const auto old_capacity = m_data.m_value.array->capacity(); + const auto old_capacity = array_capacity(); m_data.m_value.array->push_back(val); - set_parent(m_data.m_value.array->back(), old_capacity); + set_parent_after_array_insert(m_data.m_value.array->back(), old_capacity); } /// @brief add an object to an array @@ -3323,9 +3380,9 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } // add the element to the array (perfect forwarding) - const auto old_capacity = m_data.m_value.array->capacity(); + const auto old_capacity = array_capacity(); m_data.m_value.array->emplace_back(std::forward(args)...); - return set_parent(m_data.m_value.array->back(), old_capacity); + return set_parent_after_array_insert(m_data.m_value.array->back(), old_capacity); } /// @brief add an object to an object if key does not exist @@ -3404,7 +3461,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @sa https://json.nlohmann.me/api/basic_json/insert/ iterator insert(const_iterator pos, basic_json&& val) // NOLINT(performance-unnecessary-value-param) { - return insert(pos, val); + return insert(std::move(pos), val); } /// @brief inserts copies of element into array diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index d4972a35e..35443e141 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -3362,6 +3362,8 @@ NLOHMANN_JSON_NAMESPACE_END +#include // size_t + // #include @@ -3369,44 +3371,48 @@ NLOHMANN_JSON_NAMESPACE_BEGIN namespace detail { -/*! -@brief replace all occurrences of a substring by another string - -@param[in,out] s the string to manipulate; changed so that all - occurrences of @a f are replaced with @a t -@param[in] f the substring to replace with @a t -@param[in] t the string to replace @a f - -@pre The search string @a f must not be empty. **This precondition is -enforced with an assertion.** - -@since version 2.0.0 -*/ -template -inline void replace_substring(StringType& s, const StringType& f, - const StringType& t) -{ - JSON_ASSERT(!f.empty()); - for (auto pos = s.find(f); // find the first occurrence of f - pos != StringType::npos; // make sure f was found - s.replace(pos, f.size(), t), // replace with t, and - pos = s.find(f, pos + t.size())) // find the next occurrence of f - {} -} - /*! * @brief string escaping as described in RFC 6901 (Sect. 4) * @param[in] s string to escape * @return escaped string * * Note the order of escaping "~" to "~0" and "/" to "~1" is important. + * + * The string is rebuilt in a single pass, appending whole runs between the + * characters that need escaping. Scanning with find_first_of() keeps the + * common case -- nothing to escape -- as fast as a single search, while + * repeated replace() calls would move the tail of the string once per + * escaped character. */ template -inline StringType escape(StringType s) +inline StringType escape(const StringType& s) { - replace_substring(s, StringType{"~"}, StringType{"~0"}); - replace_substring(s, StringType{"/"}, StringType{"~1"}); - return s; + auto next_special = [&s](std::size_t from) + { + const auto tilde = s.find_first_of('~', from); + const auto slash = s.find_first_of('/', from); + return tilde < slash ? tilde : slash; // npos is the largest value + }; + + auto pos = next_special(0); + if (pos == StringType::npos) + { + return s; + } + + StringType result; + result.reserve(s.size() + 2); + + std::size_t run = 0; + while (pos != StringType::npos) + { + result.append(s.data() + run, pos - run); + result.append(s[pos] == '~' ? "~0" : "~1", 2); + run = pos + 1; + pos = next_special(run); + } + result.append(s.data() + run, s.size() - run); + return result; } /*! @@ -3415,12 +3421,43 @@ inline StringType escape(StringType s) * @return unescaped string * * Note the order of escaping "~1" to "/" and "~0" to "~" is important. + * + * Rebuilt in a single pass, see @ref escape. A "~" that is followed by + * neither "0" nor "1" is passed through unchanged; @ref json_pointer rejects + * such input before it gets here. */ template inline void unescape(StringType& s) { - replace_substring(s, StringType{"~1"}, StringType{"/"}); - replace_substring(s, StringType{"~0"}, StringType{"~"}); + auto pos = s.find_first_of('~', 0); + if (pos == StringType::npos) + { + return; + } + + StringType result; + result.reserve(s.size()); + + std::size_t run = 0; + while (pos != StringType::npos) + { + result.append(s.data() + run, pos - run); + + const auto next = pos + 1; + if (next < s.size() && (s[next] == '0' || s[next] == '1')) + { + result.append(s[next] == '0' ? "~" : "/", 1); + run = pos + 2; + } + else + { + result.append("~", 1); + run = pos + 1; + } + pos = s.find_first_of('~', run); + } + result.append(s.data() + run, s.size() - run); + s = result; } } // namespace detail @@ -4000,17 +4037,18 @@ struct has_to_json < BasicJsonType, T, enable_if_t < !is_basic_json::value >> template using detect_key_compare = typename T::key_compare; -template -struct has_key_compare : std::integral_constant::value> {}; - -// obtains the actual object key comparator +// obtains the actual object key comparator: object_t::key_compare if the +// object type defines it, and default_object_comparator_t otherwise +// +// note detected_or_t is used rather than std::conditional, because the latter +// names both of its type arguments eagerly; object_t::key_compare would then +// be a hard error for an object type that does not define it template struct actual_object_comparator { using object_t = typename BasicJsonType::object_t; using object_comparator_t = typename BasicJsonType::default_object_comparator_t; - using type = typename std::conditional < has_key_compare::value, - typename object_t::key_compare, object_comparator_t>::type; + using type = detected_or_t; }; template @@ -4606,6 +4644,22 @@ using has_erase_with_key_type = typename std::conditional < std::true_type, std::false_type >::type; +template +using detect_erase_with_iterator = decltype(std::declval().erase(std::declval())); + +// type trait to check if erase(iterator) returns void instead of the following +// iterator, as the object types that do not compute a successor the caller may +// not need do +template +using erase_returns_void = is_detected_exact; + +template +using detect_capacity = decltype(std::declval().capacity()); + +// type trait to check if a type has a capacity() member function +template +struct has_capacity : std::integral_constant::value> {}; + // a naive helper to check if a type is an ordered_map (exploits the fact that // ordered_map inherits capacity() from std::vector) template @@ -5013,7 +5067,10 @@ class exception : public std::exception { if (&element.second == current) { - tokens.emplace_back(element.first.c_str()); + // data() is null-terminated, so a key containing + // a null byte is cut short here rather than + // truncating the whole message at what() + tokens.emplace_back(element.first.data()); break; } } @@ -7032,7 +7089,9 @@ std::size_t hash(const BasicJsonType& j) seed = combine(seed, static_cast(j.get_binary().subtype())); for (const auto byte : j.get_binary()) { - seed = combine(seed, std::hash {}(byte)); + // the cast is needed for binary types whose value type is not + // an integer (e.g., std::byte) + seed = combine(seed, std::hash {}(static_cast(byte))); } return seed; } @@ -15131,7 +15190,10 @@ class binary_reader number_string, out_of_range::create(406, concat("number overflow parsing '", number_string, '\''), nullptr)); } - return sax->number_float(parsed_float, std::move(number_string)); + // number_string is a std::string, while the SAX interface takes a + // string_t; convert explicitly, as the two are only implicitly + // convertible for some string types + return sax->number_float(parsed_float, string_t(number_string.data(), number_string.size())); } case token_type::uninitialized: case token_type::literal_true: @@ -16325,8 +16387,13 @@ class iter_impl // NOLINT(cppcoreguidelines-special-member-functions,hicpp-speci iter_impl() = default; ~iter_impl() = default; - iter_impl(iter_impl&&) noexcept = default; - iter_impl& operator=(iter_impl&&) noexcept = default; + // the exception specification is left to be computed rather than declared: + // an array or object type whose iterator is not nothrow move constructible + // (std::deque's is not before libstdc++ 11) would make a declared noexcept + // differ from the implicit one, which deletes the function -- and is an + // error outright with older compilers + iter_impl(iter_impl&&) = default; // NOLINT(hicpp-noexcept-move,performance-noexcept-move-constructor,cppcoreguidelines-noexcept-move-operations) + iter_impl& operator=(iter_impl&&) = default; // NOLINT(hicpp-noexcept-move,performance-noexcept-move-constructor,cppcoreguidelines-noexcept-move-operations) /*! @brief constructor for a given JSON instance @@ -17207,6 +17274,7 @@ NLOHMANN_JSON_NAMESPACE_END #endif // JSON_NO_IO #include // max #include // accumulate +#include // set #include // string #include // move #include // vector @@ -17266,7 +17334,7 @@ class json_pointer string_t{}, [](const string_t& a, const string_t& b) { - return detail::concat(a, '/', detail::escape(b)); + return detail::concat(a, '/', detail::escape(b)); }); } @@ -17460,7 +17528,7 @@ class json_pointer JSON_THROW(detail::parse_error::create(109, 0, detail::concat("array index '", s, "' is not a number"), nullptr)); } - const char* p = s.c_str(); + const char* p = s.data(); char* p_end = nullptr; // NOLINT(misc-const-correctness) errno = 0; // strtoull doesn't reset errno const unsigned long long res = std::strtoull(p, &p_end, 10); // NOLINT(runtime/int) @@ -17495,19 +17563,35 @@ class json_pointer } private: + /*! + @brief the reference token sequences that denote arrays + + @ref unflatten collects the pointer prefixes that have a reference token 0 + among their children; @ref get_and_create creates arrays exactly below + those prefixes and objects everywhere else. Deciding this up front keeps + the result independent of the order in which the flattened object is + iterated, which is unspecified for some object types. + */ + using array_parents_t = std::set>; + /*! @brief create and return a reference to the pointed to value @complexity Linear in the number of reference tokens. + @throw parse_error.106 if an array index begins with '0' @throw parse_error.109 if array index is not a number @throw type_error.313 if value cannot be unflattened */ template - BasicJsonType& get_and_create(BasicJsonType& j) const + BasicJsonType& get_and_create(BasicJsonType& j, const array_parents_t& array_parents) const { auto* result = &j; + // the reference tokens that have been consumed so far; used to look up + // whether the value to be created below is an array or an object + std::vector prefix; + // in case no reference tokens exist, return a reference to the JSON value // j which will be overwritten by a primitive value for (const auto& reference_token : reference_tokens) @@ -17516,10 +17600,11 @@ class json_pointer { case detail::value_t::null: { - if (reference_token == "0") + if (array_parents.find(prefix) != array_parents.end()) { - // start a new array if the reference token is 0 - result = &result->operator[](0); + // some reference token below this position is 0, so the + // value is an array + result = &result->operator[](array_index(reference_token)); } else { @@ -17559,6 +17644,8 @@ class json_pointer default: JSON_THROW(detail::type_error::create(313, "invalid value to unflatten", &j)); } + + prefix.push_back(reference_token); } return *result; @@ -18032,7 +18119,8 @@ class json_pointer { // use the text between the beginning of the reference token // (start) and the last slash (slash). - auto reference_token = reference_string.substr(start, slash - start); + const auto count = (slash == string_t::npos ? reference_string.size() : slash) - start; + auto reference_token = string_t(reference_string.data() + start, count); // check reference tokens are properly escaped for (std::size_t pos = reference_token.find_first_of('~'); @@ -18148,6 +18236,24 @@ class json_pointer BasicJsonType result; + // collect the pointer prefixes that have a reference token 0 among + // their children; the values below them are arrays, all others are + // objects (see array_parents_t) + array_parents_t array_parents; + for (const auto& element : *value.m_data.m_value.object) + { + json_pointer ptr(element.first); + std::vector prefix; + for (auto& reference_token : ptr.reference_tokens) + { + if (reference_token == "0") + { + array_parents.insert(prefix); + } + prefix.push_back(std::move(reference_token)); + } + } + // iterate the JSON object values for (const auto& element : *value.m_data.m_value.object) { @@ -18160,7 +18266,7 @@ class json_pointer // that if the JSON pointer is "" (i.e., points to the whole value), // function get_and_create returns a reference to the result itself. // An assignment will then create a primitive value. - json_pointer(element.first).get_and_create(result) = element.second; + json_pointer(element.first).get_and_create(result, array_parents) = element.second; } return result; @@ -18832,7 +18938,7 @@ class binary_writer // step 2: write the string oa->write_characters( - reinterpret_cast(j.m_data.m_value.string->c_str()), + reinterpret_cast(j.m_data.m_value.string->data()), j.m_data.m_value.string->size()); break; } @@ -19152,7 +19258,7 @@ class binary_writer // step 2: write the string oa->write_characters( - reinterpret_cast(j.m_data.m_value.string->c_str()), + reinterpret_cast(j.m_data.m_value.string->data()), j.m_data.m_value.string->size()); break; } @@ -19369,7 +19475,7 @@ class binary_writer } write_number_with_ubjson_prefix(j.m_data.m_value.string->size(), true, use_bjdata); oa->write_characters( - reinterpret_cast(j.m_data.m_value.string->c_str()), + reinterpret_cast(j.m_data.m_value.string->data()), j.m_data.m_value.string->size()); break; } @@ -19468,7 +19574,9 @@ class binary_writer for (size_t i = 0; i < j.m_data.m_value.binary->size(); ++i) { oa->write_character(to_char_type(bjdata_draft3 ? 'B' : 'U')); - oa->write_character(to_char_type(j.m_data.m_value.binary->data()[i])); + // the cast is needed for binary types whose value type + // is not an integer (e.g., std::byte) + oa->write_character(to_char_type(static_cast(j.m_data.m_value.binary->data()[i]))); } } @@ -19529,7 +19637,7 @@ class binary_writer { write_number_with_ubjson_prefix(el.first.size(), true, use_bjdata); oa->write_characters( - reinterpret_cast(el.first.c_str()), + reinterpret_cast(el.first.data()), el.first.size()); write_ubjson(el.second, use_count, use_type, prefix_required, use_bjdata, bjdata_version); } @@ -19592,8 +19700,11 @@ class binary_writer { oa->write_character(to_char_type(element_type)); oa->write_characters( - reinterpret_cast(name.c_str()), - name.size() + 1u); + reinterpret_cast(name.data()), + name.size()); + // the terminating null byte is written explicitly rather than taken + // from the buffer, so that string_t::data() need not be null-terminated + oa->write_character(to_char_type(0x00)); } /*! @@ -19634,8 +19745,11 @@ class binary_writer write_number(to_bson_length(value.size() + 1ul), true); oa->write_characters( - reinterpret_cast(value.c_str()), - value.size() + 1); + reinterpret_cast(value.data()), + value.size()); + // the terminating null byte is written explicitly rather than taken + // from the buffer, so that string_t::data() need not be null-terminated + oa->write_character(to_char_type(0x00)); } /*! @@ -19726,7 +19840,11 @@ class binary_writer const std::size_t embedded_document_size = std::accumulate(std::begin(value), std::end(value), static_cast(0), [&array_index](std::size_t result, const typename BasicJsonType::array_t::value_type & el) { - return result + calc_bson_element_size(std::to_string(array_index++), el); + // the index is built as a std::string, while calc_bson_element_size + // takes a string_t; convert explicitly, as the two are only + // implicitly convertible for some string types + const auto key = std::to_string(array_index++); + return result + calc_bson_element_size(string_t(key.data(), key.size()), el); }); return sizeof(std::int32_t) + embedded_document_size + 1ul; @@ -19753,7 +19871,11 @@ class binary_writer for (const auto& el : value) { - write_bson_element(std::to_string(array_index++), el); + // the index is built as a std::string, while write_bson_element takes + // a string_t; convert explicitly, as the two are only implicitly + // convertible for some string types + const auto key = std::to_string(array_index++); + write_bson_element(string_t(key.data(), key.size()), el); } oa->write_character(to_char_type(0x00)); @@ -22782,7 +22904,7 @@ class serializer { case error_handler_t::strict: { - JSON_THROW(type_error::create(316, concat("incomplete UTF-8 string; last byte: 0x", hex_bytes(static_cast(s.back() | 0))), nullptr)); + JSON_THROW(type_error::create(316, concat("incomplete UTF-8 string; last byte: 0x", hex_bytes(static_cast(s[s.size() - 1] | 0))), nullptr)); } case error_handler_t::ignore: @@ -23043,6 +23165,19 @@ class serializer pos += 6; } + /*! + @brief convert a single element of a binary value to its byte value + + The elements of a binary value are dumped as the numbers 0..255, regardless + of the value type of the configured BinaryType: that type may be signed + (`char`), unsigned (`std::uint8_t`), or not an integer at all + (`std::byte`), none of which @ref dump_integer can handle uniformly. + */ + static std::uint8_t to_byte_value(binary_char_t x) noexcept + { + return static_cast(x); + } + // templates to avoid warnings about useless casts template ::value, int> = 0> bool is_negative_number(NumberType x) @@ -23064,8 +23199,10 @@ class serializer an arbitrary number, and the three digits it takes at most are written straight into the write buffer. - Any byte type that is not a plain unsigned byte is left to @ref dump_integer, - whose representation of it may differ. + Any byte type that is not a plain unsigned byte is converted to its + @ref to_byte_value "byte value" and left to @ref dump_integer, so a signed + or non-integral BinaryType::value_type (`char`, `std::byte`, ...) still + dumps as 0..255. */ template void dump_byte(const ByteType value) @@ -23078,7 +23215,7 @@ class serializer template void dump_byte(const ByteType value, std::false_type /*is_plain_byte*/) { - dump_integer(value); + dump_integer(to_byte_value(value)); } template @@ -23124,8 +23261,7 @@ class serializer template < typename NumberType, detail::enable_if_t < std::is_integral::value || std::is_same::value || - std::is_same::value || - std::is_same::value, + std::is_same::value, int > = 0 > void dump_integer(NumberType x) { @@ -24173,6 +24309,18 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @} + // Two template parameter requirements that would otherwise be silently + // violated: neither produces a diagnostic of its own, and both corrupt + // values rather than failing. + + static_assert(sizeof(typename BinaryType::value_type) == 1, + "BinaryType::value_type must be exactly one byte wide, " + "because the binary readers and writers reinterpret the container's storage as raw bytes"); + + static_assert(sizeof(NumberUnsignedType) >= sizeof(NumberIntegerType), + "NumberUnsignedType must be at least as wide as NumberIntegerType, " + "because it has to hold the absolute value of every NumberIntegerType value"); + private: /// helper for exception-safe object creation @@ -24553,21 +24701,76 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec return it; } - reference set_parent(reference j, std::size_t old_capacity = detail::unknown_size()) + /// @brief erase an element from the object and return the following one + /// Not every map returns an iterator from erase(iterator): some containers + /// (e.g., Abseil's hash maps) return void to avoid computing a successor + /// the caller may not need. Compute it before erasing for those. + template < typename It, detail::enable_if_t < + !detail::erase_returns_void::value, int > = 0 > + typename object_t::iterator erase_from_object(It pos) + { + return m_data.m_value.object->erase(pos); + } + + template < typename It, detail::enable_if_t < + detail::erase_returns_void::value, int > = 0 > + typename object_t::iterator erase_from_object(It pos) + { + auto next = std::next(pos); + m_data.m_value.object->erase(pos); + return next; + } + + /// @brief the capacity of the stored array, or unknown_size() + /// Only JSON_DIAGNOSTICS uses the value, to detect a reallocation that + /// would invalidate the parent pointers. Array types that do not have a + /// capacity() member function report unknown_size(), which is treated as + /// "the elements may have moved". +#if JSON_DIAGNOSTICS + template < typename A = array_t, detail::enable_if_t < detail::has_capacity::value, int > = 0 > + std::size_t array_capacity() const noexcept + { + return m_data.m_value.array->capacity(); + } + + template < typename A = array_t, detail::enable_if_t < !detail::has_capacity::value, int > = 0 > + std::size_t array_capacity() const noexcept + { + return detail::unknown_size(); + } +#else + static constexpr std::size_t array_capacity() noexcept + { + return detail::unknown_size(); + } +#endif + + /// @brief set the parent of a value that has just been added to an array + /// @param j the added value + /// @param old_capacity the value @ref array_capacity() returned before the + /// insertion + reference set_parent_after_array_insert(reference j, std::size_t old_capacity) { #if JSON_DIAGNOSTICS - if (old_capacity != detail::unknown_size()) + // see https://github.com/nlohmann/json/issues/2838 + JSON_ASSERT(type() == value_t::array); + if (JSON_HEDLEY_UNLIKELY(old_capacity == detail::unknown_size() + || array_capacity() != old_capacity)) { - // see https://github.com/nlohmann/json/issues/2838 - JSON_ASSERT(type() == value_t::array); - if (JSON_HEDLEY_UNLIKELY(m_data.m_value.array->capacity() != old_capacity)) - { - // capacity has changed: update all parents - set_parents(); - return j; - } + // the capacity has changed, or the array type does not let us tell: + // the elements may have moved, so update all parents + set_parents(); + return j; } +#else + static_cast(old_capacity); +#endif + return set_parent(j); + } + reference set_parent(reference j) + { +#if JSON_DIAGNOSTICS // ordered_json uses a vector internally, so pointers could have // been invalidated; see https://github.com/nlohmann/json/issues/2962 #ifdef JSON_HEDLEY_MSVC_VERSION @@ -24586,7 +24789,6 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec j.m_parent = this; #else static_cast(j); - static_cast(old_capacity); #endif return j; } @@ -25798,22 +26000,17 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec reference at(size_type idx) { // at only works for arrays - if (JSON_HEDLEY_LIKELY(is_array())) - { - JSON_TRY - { - return set_parent(m_data.m_value.array->at(idx)); - } - JSON_CATCH (std::out_of_range&) - { - // create a better exception explanation - JSON_THROW(out_of_range::create(401, detail::concat("array index ", std::to_string(idx), " is out of range"), this)); - } // cppcheck-suppress[missingReturn] - } - else + if (JSON_HEDLEY_UNLIKELY(!is_array())) { JSON_THROW(type_error::create(304, detail::concat("cannot use at() with ", type_name()), this)); } + + if (JSON_HEDLEY_UNLIKELY(idx >= m_data.m_value.array->size())) + { + JSON_THROW(out_of_range::create(401, detail::concat("array index ", std::to_string(idx), " is out of range"), this)); + } + + return set_parent((*m_data.m_value.array)[idx]); } /// @brief access specified array element with bounds checking @@ -25821,22 +26018,17 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const_reference at(size_type idx) const { // at only works for arrays - if (JSON_HEDLEY_LIKELY(is_array())) - { - JSON_TRY - { - return m_data.m_value.array->at(idx); - } - JSON_CATCH (std::out_of_range&) - { - // create a better exception explanation - JSON_THROW(out_of_range::create(401, detail::concat("array index ", std::to_string(idx), " is out of range"), this)); - } // cppcheck-suppress[missingReturn] - } - else + if (JSON_HEDLEY_UNLIKELY(!is_array())) { JSON_THROW(type_error::create(304, detail::concat("cannot use at() with ", type_name()), this)); } + + if (JSON_HEDLEY_UNLIKELY(idx >= m_data.m_value.array->size())) + { + JSON_THROW(out_of_range::create(401, detail::concat("array index ", std::to_string(idx), " is out of range"), this)); + } + + return (*m_data.m_value.array)[idx]; } /// @brief access specified object element with bounds checking @@ -25936,12 +26128,13 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec #if JSON_DIAGNOSTICS // remember array size & capacity before resizing const auto old_size = m_data.m_value.array->size(); - const auto old_capacity = m_data.m_value.array->capacity(); + const auto old_capacity = array_capacity(); #endif m_data.m_value.array->resize(idx + 1); #if JSON_DIAGNOSTICS - if (JSON_HEDLEY_UNLIKELY(m_data.m_value.array->capacity() != old_capacity)) + if (JSON_HEDLEY_UNLIKELY(old_capacity == detail::unknown_size() + || array_capacity() != old_capacity)) { // capacity has changed: update all parents set_parents(); @@ -26332,7 +26525,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec case value_t::object: { - result.m_it.object_iterator = m_data.m_value.object->erase(pos.m_it.object_iterator); + result.m_it.object_iterator = erase_from_object(pos.m_it.object_iterator); break; } @@ -26971,9 +27164,9 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } // add the element to the array (move semantics) - const auto old_capacity = m_data.m_value.array->capacity(); + const auto old_capacity = array_capacity(); m_data.m_value.array->push_back(std::move(val)); - set_parent(m_data.m_value.array->back(), old_capacity); + set_parent_after_array_insert(m_data.m_value.array->back(), old_capacity); // if val is moved from, basic_json move constructor marks it null, so we do not call the destructor } @@ -27004,9 +27197,9 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } // add the element to the array - const auto old_capacity = m_data.m_value.array->capacity(); + const auto old_capacity = array_capacity(); m_data.m_value.array->push_back(val); - set_parent(m_data.m_value.array->back(), old_capacity); + set_parent_after_array_insert(m_data.m_value.array->back(), old_capacity); } /// @brief add an object to an array @@ -27092,9 +27285,9 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } // add the element to the array (perfect forwarding) - const auto old_capacity = m_data.m_value.array->capacity(); + const auto old_capacity = array_capacity(); m_data.m_value.array->emplace_back(std::forward(args)...); - return set_parent(m_data.m_value.array->back(), old_capacity); + return set_parent_after_array_insert(m_data.m_value.array->back(), old_capacity); } /// @brief add an object to an object if key does not exist @@ -27173,7 +27366,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @sa https://json.nlohmann.me/api/basic_json/insert/ iterator insert(const_iterator pos, basic_json&& val) // NOLINT(performance-unnecessary-value-param) { - return insert(pos, val); + return insert(std::move(pos), val); } /// @brief inserts copies of element into array diff --git a/tests/src/unit-alt-string.cpp b/tests/src/unit-alt-string.cpp index 46e062c6e..ec2ac146b 100644 --- a/tests/src/unit-alt-string.cpp +++ b/tests/src/unit-alt-string.cpp @@ -11,8 +11,10 @@ #include +#include #include #include +#include /* forward declarations */ class alt_string; @@ -22,6 +24,10 @@ void int_to_string(alt_string& target, std::size_t value); // NOLINT(misc-use-in /* * This is virtually a string class. * It covers std::string under the hood. + * + * It deliberately does not provide c_str(), back(), find(str, pos), replace(), + * or substr(): the library must not rely on them. Do not add members here + * without checking that the library actually needs them. */ class alt_string { @@ -106,11 +112,6 @@ class alt_string return str_impl < op.str_impl; } - const char* c_str() const - { - return str_impl.c_str(); - } - char& operator[](std::size_t index) { return str_impl[index]; @@ -121,16 +122,6 @@ class alt_string return str_impl[index]; } - char& back() - { - return str_impl.back(); - } - - const char& back() const - { - return str_impl.back(); - } - void clear() { str_impl.clear(); @@ -146,28 +137,11 @@ class alt_string return str_impl.empty(); } - std::size_t find(const alt_string& str, std::size_t pos = 0) const - { - return str_impl.find(str.str_impl, pos); - } - std::size_t find_first_of(char c, std::size_t pos = 0) const { return str_impl.find_first_of(c, pos); } - alt_string substr(std::size_t pos = 0, std::size_t count = npos) const - { - const std::string s = str_impl.substr(pos, count); - return {s.data(), s.size()}; - } - - alt_string& replace(std::size_t pos, std::size_t count, const alt_string& str) - { - str_impl.replace(pos, count, str.str_impl); - return *this; - } - void reserve( std::size_t new_cap = 0 ) { str_impl.reserve(new_cap); @@ -202,6 +176,31 @@ bool operator<(const char* op1, const alt_string& op2) noexcept TEST_CASE("alternative string type") { + SECTION("binary formats") + { + alt_json doc; + doc["pi"] = 3.141; + doc["happy"] = true; + doc["list"] = {1, 2, 3}; + + CHECK(alt_json::from_cbor(alt_json::to_cbor(doc)) == doc); + CHECK(alt_json::from_msgpack(alt_json::to_msgpack(doc)) == doc); + // BSON is not covered: it additionally needs string_t::find(value_type), + // which alt_string does not provide + CHECK(alt_json::from_ubjson(alt_json::to_ubjson(doc)) == doc); + + // a UBJSON high-precision number is parsed into a std::string that the + // reader has to hand to the SAX interface as an alt_string + const std::vector high_precision = + { + 'H', 'i', 0x16, '3', '.', '1', '4', '1', '5', '9', '2', '6', '5', '3', + '5', '8', '9', '7', '9', '3', '2', '3', '8', '4', '6' + }; + const auto number = alt_json::from_ubjson(high_precision); + CHECK(number.is_number_float()); + CHECK(number.get() == doctest::Approx(3.14159265358979323846)); + } + SECTION("dump") { { @@ -332,6 +331,15 @@ TEST_CASE("alternative string type") CHECK(j.at(alt_json::json_pointer("/foo/0")) == j["foo"][0]); CHECK(j.at(alt_json::json_pointer("/foo/1")) == j["foo"][1]); + + // RFC 6901 escaping works without string_t::find(str, pos), replace(), + // and substr() + auto j2 = alt_json::parse(R"({"a/b": 1, "m~n": 2, "~/~~//": 3})"); + CHECK(j2.at(alt_json::json_pointer("/a~1b")) == 1); + CHECK(j2.at(alt_json::json_pointer("/m~0n")) == 2); + CHECK(j2.at(alt_json::json_pointer("/~0~1~0~0~1~1")) == 3); + CHECK(alt_json::json_pointer("/~0~1~0~0~1~1").to_string() == alt_string("/~0~1~0~0~1~1")); + CHECK(j2.flatten().unflatten() == j2); } SECTION("patch") diff --git a/tests/src/unit-custom-array-type.cpp b/tests/src/unit-custom-array-type.cpp new file mode 100644 index 000000000..00606c6e0 --- /dev/null +++ b/tests/src/unit-custom-array-type.cpp @@ -0,0 +1,150 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#include "doctest_compatibility.h" + +#include + +#include +#include +#include +#include +#include +#include + +namespace +{ + +// std::deque has no capacity() member function, which the library only needs +// to detect a reallocation for JSON_DIAGNOSTICS +using deque_json = nlohmann::basic_json; + +// a std::vector whose at() is hidden: the library performs its own bounds +// check and must not fall back to the container's checked accessor +template> +class vector_without_at : public std::vector +{ + public: + vector_without_at() = default; + + // the array of an initializer list is built from a range + template + vector_without_at(InputIt first, InputIt last) : std::vector(first, last) {} + + void at() = delete; +}; + +using no_at_json = nlohmann::basic_json; + +} // namespace + +TEST_CASE("array type without capacity()") +{ + SECTION("the iterators take their exception specification from the container") + { + // basic_json's iterators move exactly as the container iterators do: + // their move operations are defaulted without a declared noexcept, + // because an array or object type whose iterator is not nothrow move + // constructible would otherwise have them deleted (std::deque's is not + // with libstdc++ before 11, and neither are MSVC's debug iterators) + CHECK(std::is_nothrow_move_constructible::value == + (std::is_nothrow_move_constructible::value + && std::is_nothrow_move_constructible::value)); + CHECK(std::is_nothrow_move_assignable::value == + (std::is_nothrow_move_assignable::value + && std::is_nothrow_move_assignable::value)); + CHECK(std::is_nothrow_move_constructible::value == + (std::is_nothrow_move_constructible::value + && std::is_nothrow_move_constructible::value)); + + // and they are movable at all, which is what dropping the declared + // noexcept buys for a std::deque array + CHECK(std::is_move_constructible::value); + CHECK(std::is_move_assignable::value); + } + + SECTION("adding elements") + { + deque_json j = deque_json::array(); + j.push_back(1); + j.push_back("two"); + j.emplace_back(3); + j += 4; + + CHECK(j.size() == 4); + CHECK(j == deque_json({1, "two", 3, 4})); + CHECK(j.back() == 4); + CHECK(j.front() == 1); + } + + SECTION("accessing and modifying elements") + { + auto j = deque_json::parse(R"([1,2,3])"); + + CHECK(j[1] == 2); + CHECK(j.at(2) == 3); + + // growing through operator[] fills up with null values + j[5] = 6; + CHECK(j.size() == 6); + CHECK(j[4].is_null()); + CHECK(j[5] == 6); + + j.erase(0); + CHECK(j == deque_json({2, 3, nullptr, nullptr, 6})); + + auto it = j.erase(j.begin()); + CHECK(*it == 3); + + j.insert(j.begin(), 1); + CHECK(j.front() == 1); + } + + SECTION("serialization and deserialization") + { + const auto j = deque_json::parse(R"({"a":[1,[2,3]],"b":[]})"); + CHECK(j.dump() == R"({"a":[1,[2,3]],"b":[]})"); + CHECK(deque_json::parse(j.dump()) == j); + CHECK(deque_json::from_cbor(deque_json::to_cbor(j)) == j); + + // empty containers are flattened to null and cannot be restored + const auto nested = deque_json::parse(R"({"a":[1,[2,3]]})"); + CHECK(nested.flatten().unflatten() == nested); + } + + SECTION("references stay valid while the array grows") + { + deque_json j = deque_json::array(); + j.push_back(1); + auto& first = j[0]; + for (int i = 0; i < 100; ++i) + { + j.push_back(i); + } + CHECK(&first == &j[0]); + CHECK(first == 1); + } +} + +TEST_CASE("array type without at()") +{ + // built in memory rather than parsed, so that the exception message does + // not gain a byte range with JSON_DIAGNOSTIC_POSITIONS + no_at_json j = {1, 2, 3}; + const auto& jc = j; + + CHECK(j.at(0) == 1); + CHECK(j.at(2) == 3); + CHECK(jc.at(2) == 3); + + CHECK_THROWS_WITH_AS(j.at(3), "[json.exception.out_of_range.401] array index 3 is out of range", no_at_json::out_of_range); + CHECK_THROWS_WITH_AS(jc.at(3), "[json.exception.out_of_range.401] array index 3 is out of range", no_at_json::out_of_range); + + CHECK(j.at(no_at_json::json_pointer("/1")) == 2); + CHECK_THROWS_AS(j.at(no_at_json::json_pointer("/3")), no_at_json::out_of_range); +} diff --git a/tests/src/unit-custom-binary-type.cpp b/tests/src/unit-custom-binary-type.cpp new file mode 100644 index 000000000..d357ec9a3 --- /dev/null +++ b/tests/src/unit-custom-binary-type.cpp @@ -0,0 +1,79 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#include "doctest_compatibility.h" + +#include + +#include +#include +#include +#include +#include +#include + +#ifdef JSON_HAS_CPP_17 + #include +#endif + +namespace +{ + +// a BinaryType whose value type is signed: the elements must still be +// processed as the numbers 0..255 +using char_binary_json = nlohmann::basic_json < + std::map, std::vector, std::string, bool, std::int64_t, std::uint64_t, + double, std::allocator, nlohmann::adl_serializer, std::vector, void >; + +#ifdef JSON_HAS_CPP_17 + // a BinaryType whose value type is not an integer type at all + using byte_binary_json = nlohmann::basic_json < + std::map, std::vector, std::string, bool, std::int64_t, std::uint64_t, + double, std::allocator, nlohmann::adl_serializer, std::vector, void >; +#endif + +} // namespace + +TEST_CASE("binary type whose value type is not std::uint8_t") +{ + SECTION("a signed value type does not dump negative numbers") + { + const std::vector chars{'\0', '\x01', '\xFF'}; + CHECK(char_binary_json::binary(chars).dump() == R"({"bytes":[0,1,255],"subtype":null})"); + CHECK(char_binary_json::binary(chars, 42).dump() == R"({"bytes":[0,1,255],"subtype":42})"); + CHECK(char_binary_json::binary({}).dump() == R"({"bytes":[],"subtype":null})"); + } + + SECTION("the default binary type is unchanged") + { + CHECK(nlohmann::json::binary({0, 1, 255}, 42).dump() == R"({"bytes":[0,1,255],"subtype":42})"); + } + +#ifdef JSON_HAS_CPP_17 + SECTION("dumping a value type that is not an integer") + { + const std::vector bytes{std::byte{0}, std::byte{1}, std::byte{0xFF}}; + CHECK(byte_binary_json::binary(bytes).dump() == R"({"bytes":[0,1,255],"subtype":null})"); + CHECK(byte_binary_json::binary(bytes, 42).dump() == R"({"bytes":[0,1,255],"subtype":42})"); + CHECK(byte_binary_json::binary({}).dump() == R"({"bytes":[],"subtype":null})"); + } + + SECTION("hashing and the binary formats") + { + const std::vector bytes{std::byte{0}, std::byte{1}, std::byte{0xFF}}; + const auto j = byte_binary_json::binary(bytes); + + CHECK(std::hash {}(j) == std::hash {}(j)); + CHECK(byte_binary_json::from_cbor(byte_binary_json::to_cbor(j)) == j); + CHECK(byte_binary_json::from_msgpack(byte_binary_json::to_msgpack(j)) == j); + + // UBJSON has no binary type, so binary values are written as an array + CHECK(byte_binary_json::from_ubjson(byte_binary_json::to_ubjson(j)) == byte_binary_json({0, 1, 255})); + } +#endif +} diff --git a/tests/src/unit-custom-object-type.cpp b/tests/src/unit-custom-object-type.cpp new file mode 100644 index 000000000..cb2cb5ff3 --- /dev/null +++ b/tests/src/unit-custom-object-type.cpp @@ -0,0 +1,323 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#include "doctest_compatibility.h" + +#include + +#include +#include +#include +#include +#include +#include + + +namespace +{ + +// An ObjectType that does *not* define a key_compare member type, which is +// what every hash map looks like to the library. +// +// A hash map is deliberately not used here: object_t is probed for +// key_compare inside the definition of basic_json, that is, while basic_json +// is still an incomplete type, and whether a hash map can be instantiated +// with an incomplete mapped type depends on the standard library (libstdc++ 9 +// needs the size of the mapped type for its node type and rejects it). So the +// object type wraps a std::map instead of inheriting from it: an earlier +// version derived from std::map and shadowed the inherited key_compare type +// with a same-named member function, relying on ordinary member hiding to +// make key_compare unreachable as a type. MSVC 2017 (AppVeyor, /std:c++17) +// does not honor that hiding for a typename-qualified lookup performed from +// outside the class and still resolves key_compare to the base's comparator +// type, so the library's probe incorrectly found one. Composition sidesteps +// the question entirely: with no base class, there is no key_compare to find +// under any lookup rule. +template +class no_key_compare_map +{ + using map_t = std::map; + map_t data; + + public: + using key_type = typename map_t::key_type; + using mapped_type = typename map_t::mapped_type; + using value_type = typename map_t::value_type; + using size_type = typename map_t::size_type; + using allocator_type = typename map_t::allocator_type; + using iterator = typename map_t::iterator; + using const_iterator = typename map_t::const_iterator; + + // -Weffc++ asks for the member to be initialized in the member + // initialization list, which a defaulted constructor does not do; the + // exception specification a defaulted one would have carried has to be + // written out as well, or -Wnoexcept objects where the standard library + // takes noexcept(construct(...)) + no_key_compare_map() noexcept(std::is_nothrow_default_constructible::value) : data() {} + + // converting between two basic_json types builds the object from a range + template + no_key_compare_map(InputIt first, InputIt last) : data(first, last) {} + + iterator begin() noexcept + { + return data.begin(); + } + iterator end() noexcept + { + return data.end(); + } + const_iterator begin() const noexcept + { + return data.begin(); + } + const_iterator end() const noexcept + { + return data.end(); + } + const_iterator cbegin() const noexcept + { + return data.cbegin(); + } + const_iterator cend() const noexcept + { + return data.cend(); + } + + bool empty() const noexcept + { + return data.empty(); + } + size_type size() const noexcept + { + return data.size(); + } + size_type max_size() const noexcept + { + return data.max_size(); + } + void clear() noexcept + { + data.clear(); + } + + iterator find(const key_type& key) + { + return data.find(key); + } + const_iterator find(const key_type& key) const + { + return data.find(key); + } + size_type count(const key_type& key) const + { + return data.count(key); + } + + std::pair emplace(const key_type& key, const mapped_type& value) + { + return data.emplace(key, value); + } + + std::pair insert(const value_type& value) + { + return data.insert(value); + } + + template + void insert(InputIt first, InputIt last) + { + data.insert(first, last); + } + + mapped_type& operator[](const key_type& key) + { + return data[key]; + } + + mapped_type& at(const key_type& key) + { + return data.at(key); + } + const mapped_type& at(const key_type& key) const + { + return data.at(key); + } + + iterator erase(iterator pos) + { + return data.erase(pos); + } + iterator erase(iterator first, iterator last) + { + return data.erase(first, last); + } + size_type erase(const key_type& key) + { + return data.erase(key); + } + + void swap(no_key_compare_map& other) noexcept(noexcept(data.swap(other.data))) + { + data.swap(other.data); + } + + friend bool operator==(const no_key_compare_map& lhs, const no_key_compare_map& rhs) + { + return lhs.data == rhs.data; + } + friend bool operator<(const no_key_compare_map& lhs, const no_key_compare_map& rhs) + { + return lhs.data < rhs.data; + } +}; + +using no_key_compare_json = nlohmann::basic_json; + +// An ObjectType whose erase(iterator) returns void rather than the following +// iterator, as for instance Abseil's hash maps do +template +struct void_erase_map : std::map +{ + using base_t = std::map; + using iterator = typename base_t::iterator; + using base_t::erase; + + void erase(iterator pos) + { + base_t::erase(pos); + } +}; + +using void_erase_json = nlohmann::basic_json; + +} // namespace + +TEST_CASE("object type whose erase() returns void") +{ + SECTION("erasing every element through the returned iterator") + { + void_erase_json j; + for (int i = 0; i < 8; ++i) + { + j["k" + std::to_string(i)] = i; + } + + std::size_t erased = 0; + for (auto it = j.begin(); it != j.end(); ++erased) + { + it = j.erase(it); + } + CHECK(erased == 8); + CHECK(j.empty()); + } + + SECTION("erasing in the middle returns the following element") + { + void_erase_json j; + for (int i = 0; i < 4; ++i) + { + j["k" + std::to_string(i)] = i; + } + + auto it = j.begin(); + ++it; + const auto after = j.erase(it); + CHECK(j.size() == 3); + CHECK(after.key() == "k2"); + CHECK(after.value() == 2); + CHECK(!j.contains("k1")); + } + + SECTION("the other erase overloads are unaffected") + { + void_erase_json j; + j["a"] = 1; + j["b"] = 2; + j["c"] = 3; + + CHECK(j.erase("a") == 1); + CHECK(j.erase("nope") == 0); + j.erase(j.begin(), j.end()); + CHECK(j.empty()); + } +} + +TEST_CASE("object type without key_compare") +{ + SECTION("object_comparator_t falls back to default_object_comparator_t") + { + CHECK(std::is_same < no_key_compare_json::object_comparator_t, + no_key_compare_json::default_object_comparator_t >::value); + } + + SECTION("object types defining key_compare are unaffected") + { + CHECK(std::is_same::value); + CHECK(std::is_same::value); + } + + SECTION("creating and accessing values") + { + no_key_compare_json j; + j["one"] = 1; + j["two"] = "zwei"; + j["three"]["nested"] = true; + + CHECK(j.size() == 3); + CHECK(j.at("one") == 1); + CHECK(j["two"] == "zwei"); + CHECK(j["three"]["nested"] == true); + CHECK(j.contains("one")); + CHECK(!j.contains("four")); + CHECK(j.find("one") != j.end()); + CHECK(j.count("one") == 1); + CHECK(j.erase("one") == 1); + CHECK(j.size() == 2); + } + + SECTION("serialization and deserialization") + { + const auto j = no_key_compare_json::parse(R"({"a":[1,2,3],"b":{"c":null}})"); + CHECK(j["a"].size() == 3); + CHECK(j["a"][2] == 3); + CHECK(j["b"]["c"].is_null()); + CHECK(no_key_compare_json::parse(j.dump()) == j); + } + + SECTION("binary formats") + { + const auto j = no_key_compare_json::parse(R"({"a":[1,2,3],"b":"x"})"); + CHECK(no_key_compare_json::from_cbor(no_key_compare_json::to_cbor(j)) == j); + CHECK(no_key_compare_json::from_msgpack(no_key_compare_json::to_msgpack(j)) == j); + } + + SECTION("flatten and unflatten") + { + // "o" has a key that looks like an array index, so unflatten() must + // not turn it into an array + const auto j = no_key_compare_json::parse( + R"({"c":[1,2,3],"d":{"e":"s"},"n":[[0,1],[2]],"o":{"2":"x"}})"); + CHECK(j.flatten().unflatten() == j); + } + + SECTION("conversion to and from nlohmann::json") + { + const auto j = no_key_compare_json::parse(R"({"a":1,"b":[true,null]})"); + const nlohmann::json converted(j); + + CHECK(converted.is_object()); + CHECK(converted["a"] == 1); + CHECK(converted["b"][0] == true); + CHECK(converted["b"][1].is_null()); + CHECK(no_key_compare_json(converted) == j); + } +} + diff --git a/tests/src/unit-json_pointer.cpp b/tests/src/unit-json_pointer.cpp index 4082de45c..b01df2921 100644 --- a/tests/src/unit-json_pointer.cpp +++ b/tests/src/unit-json_pointer.cpp @@ -507,6 +507,16 @@ TEST_CASE("JSON pointers") // explicit roundtrip check CHECK(j.flatten().unflatten() == j); + // an object is only unflattened to an array if one of its keys is the + // reference token 0; this must not depend on which key is seen first + CHECK(json({{"/2", "x"}}).unflatten() == json({{"2", "x"}})); + CHECK(json({{"/10", "y"}, {"/2", "z"}}).unflatten() == json({{"10", "y"}, {"2", "z"}})); + CHECK(json({{"/0", 1}, {"/1", 2}}).unflatten() == json({1, 2})); + CHECK(json({{"/1", 2}, {"/0", 1}}).unflatten() == json({1, 2})); + CHECK(json({{"/0", 1}, {"/2", 3}}).unflatten() == json({1, nullptr, 3})); + CHECK(json({{"/a/1", 2}, {"/a/0", 1}}).unflatten() == json({{"a", {1, 2}}})); + CHECK(json({{"/a/1", 2}, {"/a/x", 1}}).unflatten() == json({{"a", {{"1", 2}, {"x", 1}}}})); + // roundtrip for primitive values json j_null; CHECK(j_null.flatten().unflatten() == j_null); From af91eee2cc770a369d578273cca7ea277008f82b Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Wed, 16 Sep 2026 20:11:06 +0200 Subject: [PATCH 03/64] Bump the codeql-action group with 4 updates (#5536) Bumps the codeql-action group with 4 updates: [github/codeql-action/init](https://github.com/github/codeql-action), [github/codeql-action/autobuild](https://github.com/github/codeql-action), [github/codeql-action/analyze](https://github.com/github/codeql-action) and [github/codeql-action/upload-sarif](https://github.com/github/codeql-action). Updates `github/codeql-action/init` from 4.37.9 to 4.38.0 - [Release notes](https://github.com/github/codeql-action/releases) - [Changelog](https://github.com/github/codeql-action/blob/main/CHANGELOG.md) - [Commits](https://github.com/github/codeql-action/compare/cdf488f595d80d6e07e03d4674febd5ab45fa938...b96794f015dfd88f77b49b1c93e0fa7110f94c63) Updates `github/codeql-action/autobuild` from 4.37.9 to 4.38.0 - [Release notes](https://github.com/github/codeql-action/releases) - [Changelog](https://github.com/github/codeql-action/blob/main/CHANGELOG.md) - [Commits](https://github.com/github/codeql-action/compare/cdf488f595d80d6e07e03d4674febd5ab45fa938...b96794f015dfd88f77b49b1c93e0fa7110f94c63) Updates `github/codeql-action/analyze` from 4.37.9 to 4.38.0 - [Release notes](https://github.com/github/codeql-action/releases) - [Changelog](https://github.com/github/codeql-action/blob/main/CHANGELOG.md) - [Commits](https://github.com/github/codeql-action/compare/cdf488f595d80d6e07e03d4674febd5ab45fa938...b96794f015dfd88f77b49b1c93e0fa7110f94c63) Updates `github/codeql-action/upload-sarif` from 4.37.9 to 4.38.0 - [Release notes](https://github.com/github/codeql-action/releases) - [Changelog](https://github.com/github/codeql-action/blob/main/CHANGELOG.md) - [Commits](https://github.com/github/codeql-action/compare/cdf488f595d80d6e07e03d4674febd5ab45fa938...b96794f015dfd88f77b49b1c93e0fa7110f94c63) --- updated-dependencies: - dependency-name: github/codeql-action/init dependency-version: 4.38.0 dependency-type: direct:production update-type: version-update:semver-minor dependency-group: codeql-action - dependency-name: github/codeql-action/autobuild dependency-version: 4.38.0 dependency-type: direct:production update-type: version-update:semver-minor dependency-group: codeql-action - dependency-name: github/codeql-action/analyze dependency-version: 4.38.0 dependency-type: direct:production update-type: version-update:semver-minor dependency-group: codeql-action - dependency-name: github/codeql-action/upload-sarif dependency-version: 4.38.0 dependency-type: direct:production update-type: version-update:semver-minor dependency-group: codeql-action ... Signed-off-by: dependabot[bot] Co-authored-by: dependabot[bot] <49699333+dependabot[bot]@users.noreply.github.com> --- .github/workflows/codeql-analysis.yml | 6 +++--- .github/workflows/flawfinder.yml | 2 +- .github/workflows/scorecards.yml | 2 +- .github/workflows/semgrep.yml | 2 +- 4 files changed, 6 insertions(+), 6 deletions(-) diff --git a/.github/workflows/codeql-analysis.yml b/.github/workflows/codeql-analysis.yml index c5497ef9c..1c2e73cd1 100644 --- a/.github/workflows/codeql-analysis.yml +++ b/.github/workflows/codeql-analysis.yml @@ -38,14 +38,14 @@ jobs: # Initializes the CodeQL tools for scanning. - name: Initialize CodeQL - uses: github/codeql-action/init@cdf488f595d80d6e07e03d4674febd5ab45fa938 # v4.37.9 + uses: github/codeql-action/init@b96794f015dfd88f77b49b1c93e0fa7110f94c63 # v4.38.0 with: languages: c-cpp # Autobuild attempts to build any compiled languages (C/C++, C#, or Java). # If this step fails, then you should remove it and run the build manually (see below) - name: Autobuild - uses: github/codeql-action/autobuild@cdf488f595d80d6e07e03d4674febd5ab45fa938 # v4.37.9 + uses: github/codeql-action/autobuild@b96794f015dfd88f77b49b1c93e0fa7110f94c63 # v4.38.0 - name: Perform CodeQL Analysis - uses: github/codeql-action/analyze@cdf488f595d80d6e07e03d4674febd5ab45fa938 # v4.37.9 + uses: github/codeql-action/analyze@b96794f015dfd88f77b49b1c93e0fa7110f94c63 # v4.38.0 diff --git a/.github/workflows/flawfinder.yml b/.github/workflows/flawfinder.yml index 7c4d22b3c..2c8befd40 100644 --- a/.github/workflows/flawfinder.yml +++ b/.github/workflows/flawfinder.yml @@ -43,6 +43,6 @@ jobs: output: 'flawfinder_results.sarif' - name: Upload analysis results to GitHub Security tab - uses: github/codeql-action/upload-sarif@cdf488f595d80d6e07e03d4674febd5ab45fa938 # v4.37.9 + uses: github/codeql-action/upload-sarif@b96794f015dfd88f77b49b1c93e0fa7110f94c63 # v4.38.0 with: sarif_file: ${{github.workspace}}/flawfinder_results.sarif diff --git a/.github/workflows/scorecards.yml b/.github/workflows/scorecards.yml index de919b1ab..113da079a 100644 --- a/.github/workflows/scorecards.yml +++ b/.github/workflows/scorecards.yml @@ -76,6 +76,6 @@ jobs: # Upload the results to GitHub's code scanning dashboard. - name: "Upload to code-scanning" - uses: github/codeql-action/upload-sarif@cdf488f595d80d6e07e03d4674febd5ab45fa938 # v4.37.9 + uses: github/codeql-action/upload-sarif@b96794f015dfd88f77b49b1c93e0fa7110f94c63 # v4.38.0 with: sarif_file: results.sarif diff --git a/.github/workflows/semgrep.yml b/.github/workflows/semgrep.yml index 38de00932..dc326db55 100644 --- a/.github/workflows/semgrep.yml +++ b/.github/workflows/semgrep.yml @@ -61,7 +61,7 @@ jobs: # Upload SARIF file generated in previous step - name: Upload SARIF file - uses: github/codeql-action/upload-sarif@cdf488f595d80d6e07e03d4674febd5ab45fa938 # v4.37.9 + uses: github/codeql-action/upload-sarif@b96794f015dfd88f77b49b1c93e0fa7110f94c63 # v4.38.0 with: sarif_file: semgrep.sarif if: always() From ff65f688f723e8aea56905f77fdcf2c0761bf779 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Wed, 16 Sep 2026 20:11:19 +0200 Subject: [PATCH 04/64] Cover binary values in the indentation regression tests (#5533) #5285's indentation regression test didn't exercise json::binary, which serializes as an object but always writes its byte array compactly (dump_byte()). #5186 had covered this case before it was closed as superseded; port just that coverage here. Signed-off-by: Niels Lohmann --- tests/src/unit-serialization.cpp | 13 +++++++++++++ 1 file changed, 13 insertions(+) diff --git a/tests/src/unit-serialization.cpp b/tests/src/unit-serialization.cpp index 511108c64..00b305a75 100644 --- a/tests/src/unit-serialization.cpp +++ b/tests/src/unit-serialization.cpp @@ -522,6 +522,19 @@ TEST_CASE("indentation is written straight into the write buffer") CHECK(json::parse(out) == j); } + SECTION("binary values are indented the same way") + { + // a binary value is serialized as an object with "bytes" and + // "subtype" keys; the byte array itself is always written compactly + // (see dump_byte()), so only the surrounding object's indentation + // goes through put_indent() + const json j = json::binary({1, 2, 3}, 128); + CHECK(j.dump(2000) == "{\n" + std::string(2000, ' ') + "\"bytes\": [1, 2, 3],\n" + + std::string(2000, ' ') + "\"subtype\": 128\n}"); + CHECK(j.dump(2000, '\t') == "{\n" + std::string(2000, '\t') + "\"bytes\": [1, 2, 3],\n" + + std::string(2000, '\t') + "\"subtype\": 128\n}"); + } + SECTION("indentation is unchanged for ordinary widths") { const json j = {{"a", {1, 2}}, {"b", nullptr}}; From f58db1c9e81b2c03cb97152657b513bd53b50dac Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Wed, 16 Sep 2026 20:15:20 +0200 Subject: [PATCH 05/64] Add test coverage for ordered_json/alt_json across binary formats and patch/diff/flatten APIs (#5480) * Add test coverage for ordered_json/alt_json across binary formats and patch/diff/flatten APIs Closes a test-coverage gap from #5421: ordered_json (and the alt_string-based basic_json specialization from unit-alt-string.cpp) were never round-tripped through the binary formats (CBOR/MessagePack/UBJSON/BSON/BJData), nor through flatten()/unflatten(), diff()/patch()/patch_inplace(), or merge_patch(). Also adds a std::formatter spot-check, mirroring the precedent set by the format_as() ADL-deduction test. Signed-off-by: Niels Lohmann * Pass alt_string's std::string constructor argument by value (clang-tidy modernize-pass-by-value) Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- tests/src/unit-ordered_json2.cpp | 489 +++++++++++++++++++++++++++++++ tests/src/unit-std-format.cpp | 13 + 2 files changed, 502 insertions(+) create mode 100644 tests/src/unit-ordered_json2.cpp diff --git a/tests/src/unit-ordered_json2.cpp b/tests/src/unit-ordered_json2.cpp new file mode 100644 index 000000000..83eb7668d --- /dev/null +++ b/tests/src/unit-ordered_json2.cpp @@ -0,0 +1,489 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-FileCopyrightText: 2018 Vitaliy Manushkin +// SPDX-License-Identifier: MIT + +// This file closes a test-coverage gap described in GitHub issue #5421: +// nlohmann::ordered_json (and other non-default basic_json specializations, +// such as the alt_string-based one from unit-alt-string.cpp) were never +// exercised through the binary formats (CBOR/MessagePack/UBJSON/BSON/BJData) +// or through flatten()/unflatten()/diff()/patch()/merge_patch(). + +#include "doctest_compatibility.h" + +#include + +#include +#include +#include +#include + +using nlohmann::json; +using nlohmann::ordered_json; + +///////////////////////////////////////////////////////////////////////////// +// alt_json: a second, independent copy of the custom-string_t basic_json +// specialization defined in unit-alt-string.cpp. +// +// It is duplicated here (rather than shared via a header) because every +// unit-*.cpp file in this test suite is compiled into its own standalone +// executable (see tests/CMakeLists.txt), so there is no ODR concern in +// having the same class name defined in multiple translation units. +// +// Two members had to be added relative to the original alt_string +// (a constructor from std::string, and a find(char, pos) overload) because +// the original type was never used with the binary writers/readers before +// this file: BSON's array/document writer converts std::to_string() results +// and checks for embedded NUL characters via find(char), and the UBJSON/BSON +// high-precision-number path constructs the SAX string_t argument from a +// std::string. Neither path is exercised anywhere else in the test suite for +// this type, which is presumably why the gap was never noticed. +///////////////////////////////////////////////////////////////////////////// + +class alt_string; +bool operator<(const char* op1, const alt_string& op2) noexcept; // NOLINT(misc-use-internal-linkage) +void int_to_string(alt_string& target, std::size_t value); // NOLINT(misc-use-internal-linkage) + +class alt_string +{ + public: + using value_type = std::string::value_type; + + static constexpr auto npos = (std::numeric_limits::max)(); + + alt_string(const char* str): str_impl(str) {} + alt_string(const char* str, std::size_t count): str_impl(str, count) {} + alt_string(std::string str): str_impl(std::move(str)) {} + alt_string(size_t count, char chr): str_impl(count, chr) {} + alt_string() = default; + + alt_string& append(char ch) + { + str_impl.push_back(ch); + return *this; + } + + alt_string& append(const alt_string& str) + { + str_impl.append(str.str_impl); + return *this; + } + + alt_string& append(const char* s, std::size_t length) + { + str_impl.append(s, length); + return *this; + } + + void push_back(char c) + { + str_impl.push_back(c); + } + + template + bool operator==(const op_type& op) const + { + return str_impl == op; + } + + bool operator==(const alt_string& op) const + { + return str_impl == op.str_impl; + } + + template + bool operator!=(const op_type& op) const + { + return str_impl != op; + } + + bool operator!=(const alt_string& op) const + { + return str_impl != op.str_impl; + } + + std::size_t size() const noexcept + { + return str_impl.size(); + } + + void resize(std::size_t n) + { + str_impl.resize(n); + } + + void resize(std::size_t n, char c) + { + str_impl.resize(n, c); + } + + template + bool operator<(const op_type& op) const noexcept + { + return str_impl < op; + } + + bool operator<(const alt_string& op) const noexcept + { + return str_impl < op.str_impl; + } + + const char* c_str() const + { + return str_impl.c_str(); + } + + char& operator[](std::size_t index) + { + return str_impl[index]; + } + + const char& operator[](std::size_t index) const + { + return str_impl[index]; + } + + char& back() + { + return str_impl.back(); + } + + const char& back() const + { + return str_impl.back(); + } + + void clear() + { + str_impl.clear(); + } + + const value_type* data() const + { + return str_impl.data(); + } + + bool empty() const + { + return str_impl.empty(); + } + + std::size_t find(const alt_string& str, std::size_t pos = 0) const + { + return str_impl.find(str.str_impl, pos); + } + + // needed by binary_writer's BSON support, which probes string keys for + // embedded NUL characters via find(char) + std::size_t find(char c, std::size_t pos = 0) const + { + return str_impl.find(c, pos); + } + + std::size_t find_first_of(char c, std::size_t pos = 0) const + { + return str_impl.find_first_of(c, pos); + } + + alt_string substr(std::size_t pos = 0, std::size_t count = npos) const + { + const std::string s = str_impl.substr(pos, count); + return {s.data(), s.size()}; + } + + alt_string& replace(std::size_t pos, std::size_t count, const alt_string& str) + { + str_impl.replace(pos, count, str.str_impl); + return *this; + } + + void reserve(std::size_t new_cap = 0) + { + str_impl.reserve(new_cap); + } + + private: + std::string str_impl {}; // NOLINT(readability-redundant-member-init) + + friend bool operator<(const char* /*op1*/, const alt_string& /*op2*/) noexcept; +}; + +void int_to_string(alt_string& target, std::size_t value) +{ + target = std::to_string(value).c_str(); +} + +using alt_json = nlohmann::basic_json < + std::map, + std::vector, + alt_string, + bool, + std::int64_t, + std::uint64_t, + double, + std::allocator, + nlohmann::adl_serializer >; + +bool operator<(const char* op1, const alt_string& op2) noexcept +{ + return op1 < op2.str_impl; +} + +namespace +{ + +// collects the object keys of j, in iteration order +std::vector collect_keys(const ordered_json& j) +{ + std::vector result; + for (auto it = j.cbegin(); it != j.cend(); ++it) + { + result.push_back(it.key()); + } + return result; +} + +// a nested object/array value with keys inserted in non-alphabetical order, +// used to check both round-trip equality and (for ordered_json) that +// insertion order survives a trip through a binary format +ordered_json make_rich_ordered_json() +{ + ordered_json j; + j["zebra"] = 1; + j["apple"] = ordered_json::array({1, 2, 3}); + j["mango"]["z_nested"] = true; + j["mango"]["a_nested"] = nullptr; + j["banana"] = "some text"; + j["cherry"] = 3.14; + return j; +} + +alt_json make_rich_alt_json() +{ + alt_json j; + j["zebra"] = 1; + j["apple"] = alt_json::array({1, 2, 3}); + j["mango"]["z_nested"] = true; + j["mango"]["a_nested"] = nullptr; + j["banana"] = "some text"; + j["cherry"] = 3.14; + return j; +} + +} // namespace + +TEST_CASE("ordered_json across binary formats") +{ + const ordered_json original = make_rich_ordered_json(); + const std::vector original_keys = collect_keys(original); + const std::vector original_mango_keys = collect_keys(original["mango"]); + + SECTION("CBOR") + { + const auto bytes = ordered_json::to_cbor(original); + const auto restored = ordered_json::from_cbor(bytes); + CHECK(restored == original); + CHECK(collect_keys(restored) == original_keys); + CHECK(collect_keys(restored["mango"]) == original_mango_keys); + } + + SECTION("MessagePack") + { + const auto bytes = ordered_json::to_msgpack(original); + const auto restored = ordered_json::from_msgpack(bytes); + CHECK(restored == original); + CHECK(collect_keys(restored) == original_keys); + CHECK(collect_keys(restored["mango"]) == original_mango_keys); + } + + SECTION("UBJSON") + { + const auto bytes = ordered_json::to_ubjson(original); + const auto restored = ordered_json::from_ubjson(bytes); + CHECK(restored == original); + CHECK(collect_keys(restored) == original_keys); + CHECK(collect_keys(restored["mango"]) == original_mango_keys); + } + + SECTION("BSON") + { + const auto bytes = ordered_json::to_bson(original); + const auto restored = ordered_json::from_bson(bytes); + CHECK(restored == original); + CHECK(collect_keys(restored) == original_keys); + CHECK(collect_keys(restored["mango"]) == original_mango_keys); + } + + SECTION("BJData") + { + const auto bytes = ordered_json::to_bjdata(original); + const auto restored = ordered_json::from_bjdata(bytes); + CHECK(restored == original); + CHECK(collect_keys(restored) == original_keys); + CHECK(collect_keys(restored["mango"]) == original_mango_keys); + } +} + +TEST_CASE("alt_json (custom string_t) across binary formats") +{ + const alt_json original = make_rich_alt_json(); + + SECTION("CBOR") + { + const auto bytes = alt_json::to_cbor(original); + const auto restored = alt_json::from_cbor(bytes); + CHECK(restored == original); + } + + SECTION("MessagePack") + { + const auto bytes = alt_json::to_msgpack(original); + const auto restored = alt_json::from_msgpack(bytes); + CHECK(restored == original); + } + + SECTION("UBJSON") + { + const auto bytes = alt_json::to_ubjson(original); + const auto restored = alt_json::from_ubjson(bytes); + CHECK(restored == original); + } + + SECTION("BSON") + { + const auto bytes = alt_json::to_bson(original); + const auto restored = alt_json::from_bson(bytes); + CHECK(restored == original); + } + + SECTION("BJData") + { + const auto bytes = alt_json::to_bjdata(original); + const auto restored = alt_json::from_bjdata(bytes); + CHECK(restored == original); + } +} + +TEST_CASE("ordered_json operator== is sensitive to key order") +{ + // Unlike nlohmann::json (whose object_t is a std::map, so equality never + // depends on insertion order), ordered_json's object_t (ordered_map) is a + // std::vector> under the hood, and does not define its + // own operator==: it inherits std::vector's element-wise comparison. As a + // result, two ordered_json objects holding the very same key/value pairs + // in different insertion order compare *unequal*. This is the property + // that makes the round-trip `CHECK(restored == original)` checks above a + // meaningful order-preservation check by themselves (the explicit + // collect_keys() comparisons make that check explicit/readable, and + // guard against this operator== behavior ever changing). + ordered_json a; + a["x"] = 1; + a["y"] = 2; + + ordered_json b; + b["y"] = 2; + b["x"] = 1; + + CHECK(a.size() == b.size()); + CHECK(a["x"] == b["x"]); + CHECK(a["y"] == b["y"]); + CHECK_FALSE(a == b); +} + +TEST_CASE("duplicate keys in a binary-encoded object") +{ + // CBOR encoding of a map with two entries under the same key "a": {"a": 1, "a": 2} + const std::vector cbor_bytes + { + 0xA2, 0x61, 'a', 0x01, 0x61, 'a', 0x02 + }; + + // Both json (std::map, via operator[]) and ordered_json (ordered_map, via + // operator[]) build binary-decoded objects by looking up/creating the + // entry for each incoming key and then assigning the value into it. This + // means a repeated key does *not* produce two entries in either case; + // instead, the *first* occurrence's position is kept (relevant only for + // ordered_json) while the *last* occurrence's value wins (for both) -- + // this matches operator[]'s "assign the referenced slot" semantics, and + // is worth noting because it differs from the initializer-list + // construction path (`ordered_json{{"a",1},{"a",2}}`), which builds + // through insert()/emplace() and therefore keeps the *first* value, not + // the last (see the "There are no dup keys..." case in + // unit-ordered_json.cpp). + const auto j = json::from_cbor(cbor_bytes); + const auto oj = ordered_json::from_cbor(cbor_bytes); + + CHECK(j.size() == 1); + CHECK(oj.size() == 1); + CHECK(j["a"] == 2); + CHECK(oj["a"] == 2); + CHECK(j == json(oj)); +} + +TEST_CASE("ordered_json through flatten/unflatten") +{ + const ordered_json original = make_rich_ordered_json(); + const std::vector original_keys = collect_keys(original); + const std::vector original_mango_keys = collect_keys(original["mango"]); + + const ordered_json flat = original.flatten(); + const ordered_json unflattened = flat.unflatten(); + + CHECK(unflattened == original); + // flatten() walks the value depth-first in iteration order and + // unflatten() re-inserts each flattened key via operator[] in the flat + // object's iteration order, so for ordered_json the original key order + // (both top-level and nested) is preserved end-to-end. + CHECK(collect_keys(unflattened) == original_keys); + CHECK(collect_keys(unflattened["mango"]) == original_mango_keys); +} + +TEST_CASE("ordered_json through diff/patch/patch_inplace") +{ + ordered_json original; + original["one"] = 1; + original["two"] = 2; + original["three"] = 3; + + ordered_json target = original; + target["one"] = 100; // replace + target.erase("two"); // remove + target["four"] = 4; // add + + const ordered_json patch = ordered_json::diff(original, target); + + SECTION("patch") + { + const ordered_json patched = original.patch(patch); + CHECK(patched == target); + } + + SECTION("patch_inplace") + { + ordered_json copy = original; + copy.patch_inplace(patch); + CHECK(copy == target); + } +} + +TEST_CASE("ordered_json through merge_patch") +{ + ordered_json original; + original["a"] = 1; + original["b"] = 2; + + const ordered_json patch = {{"b", nullptr}, {"c", 3}}; + + original.merge_patch(patch); + + ordered_json expected; + expected["a"] = 1; + expected["c"] = 3; + + CHECK(original == expected); + CHECK(collect_keys(original) == collect_keys(expected)); +} diff --git a/tests/src/unit-std-format.cpp b/tests/src/unit-std-format.cpp index f7a364884..d18ccbd32 100644 --- a/tests/src/unit-std-format.cpp +++ b/tests/src/unit-std-format.cpp @@ -17,6 +17,7 @@ #include using json = nlohmann::json; +using ordered_json = nlohmann::ordered_json; // JSON_HAS_CPP_20 (do not remove; see note at top of file) #if JSON_HAS_STD_FORMAT @@ -93,4 +94,16 @@ TEST_CASE("std::formatter") } } +TEST_CASE("std::formatter") +{ + // spot-check a non-default basic_json instantiation, since the formatter + // is written against the generic NLOHMANN_BASIC_JSON_TPL_DECLARATION + // template and must actually instantiate (and behave correctly) for + // template arguments other than nlohmann::json + const ordered_json j = {{"foo", 1}, {"bar", {1, 2, 3}}}; + CHECK(std::format("{}", j) == j.dump()); + CHECK(std::format("{:#}", j) == j.dump(4)); + CHECK(std::format("{:2}", j) == j.dump(2)); +} + #endif From 75efd6b1c316c68b7240876cd536fed038b9560e Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Wed, 16 Sep 2026 20:15:46 +0200 Subject: [PATCH 06/64] Test suite: cover untested macro configs, std::formatter branches, patch_inplace, and fix a duplicate TEST_CASE name (#5492) * test: cover JSON_NO_IO, JSON_THROW/TRY/CATCH_USER, JSON_SKIP_LIBRARY_VERSION_CHECK, and JSON_DisableEnumSerialization in CI (#5423) These four supported configuration macros were never actually compiled anywhere in the test matrix: - JSON_NO_IO and the JSON_THROW_USER/JSON_TRY_USER/JSON_CATCH_USER trio are exercised together in a new tests/src/unit-no_io_and_user_exceptions.cpp, which is automatically picked up by the existing unit-*.cpp test glob and thus built across the whole standard test matrix. - JSON_SKIP_LIBRARY_VERSION_CHECK is exercised by a new, dedicated tests/src/skip_library_version_check.cpp, compiled directly by the new ci_test_skiplibraryversioncheck target in cmake/ci.cmake: the scenario it simulates (mixing two differently-versioned inclusions of the library) unavoidably triggers the compiler's own "macro redefined" warning, which would fail under the library's own -Weverything/-Werror unit test matrix for a reason unrelated to the macro under test. - JSON_DisableEnumSerialization already had #if-guarded tests in several unit-*.cpp files (from #4384), but no CMake target ever actually set the JSON_DisableEnumSerialization CMake option, so that guarded code was never compiled. Add ci_test_disableenumserialization, mirroring the existing ci_test_noimplicitconversions/ci_test_noglobaludls targets. Building the full test suite with this option on surfaced one real, narrow gap: get() on std::vector (used by unit-regression2.cpp's custom BinaryType tests) relies on std::byte being handled via enum serialization, so add the same #if-guard convention to the two affected SECTIONs there. Both new CI targets are added to the ci_cmake_options matrix in .github/workflows/ubuntu.yml, alongside the existing ci_test_* targets. Signed-off-by: Niels Lohmann * test: cover multi-digit widths and bare alignment in std::formatter (#5423) Every existing std::formatter spec with a width used a single digit (e.g. "{:2}"), so the width-parsing loop's accumulation of a second/third digit was never exercised; add multi-digit width cases. Likewise, every existing spec with an alignment character also had an explicit fill character, so the bare-alignment branch (e.g. "{:<}", with no fill) was never exercised; add cases asserting it keeps the default space indent character. Signed-off-by: Niels Lohmann * test: add coverage for patch_inplace() (#5423) patch_inplace() had no unit test at all. Add a happy-path case mirroring an existing patch() example, and -- more importantly -- pin its distinguishing contract versus patch(): when a multi-operation JSON Patch fails partway through, patch_inplace() (which mutates the document directly, operation by operation) leaves whatever operations already succeeded applied, whereas patch() (which applies the patch to an internal copy that is discarded on exception) leaves the original completely untouched either way. Verified empirically against the current implementation before writing the assertions. Signed-off-by: Niels Lohmann * test: fix duplicate TEST_CASE name in unit-no-mem-leak-on-adl-serialize.cpp (#5423) Two distinct TEST_CASEs were both named "check_for_mem_leak_on_adl_to_json-2". doctest allows duplicate names, so both still ran, but it makes --test-case= filtering and reporting ambiguous. Rename the second one to "-3", continuing the existing "-1"/"-2" sequence. Signed-off-by: Niels Lohmann * test: add direct coverage for the std::u8string to_json overload (#5423) The ADL to_json overload for std::basic_string was only ever reached indirectly, via std::filesystem::path::u8string(). Add a test that constructs a json value directly from a std::u8string, gated the same way as the overload itself (include/nlohmann/detail/conversions/to_json.hpp): behind both the std::filesystem::path feature guard and __cpp_lib_char8_t, since the overload only exists when both are satisfied. Signed-off-by: Niels Lohmann * test: verify move semantics of byte_container_with_subtype's rvalue constructors (#5423) The two rvalue-reference constructors were never distinguished from their const-lvalue-reference twins by any test. Add a "move semantics" section that constructs from an rvalue std::vector, checks the resulting container keeps the exact same buffer address as the source (a stronger check than just observing the source ended up empty, since a copy-then-clear could do that too), and confirms the source vector was left empty. Signed-off-by: Niels Lohmann * Guard patch_inplace() partial-application test against JSON_NOEXCEPTION The "distinguishing contract vs patch(): partial application on failure" test relies on doc.patch_inplace(patch) actually throwing so the partially-applied state can be observed right after the throw point. Under ci_test_noexceptions, JSON_THROW() calls std::abort() instead of throwing, and doctest's --no-throw test filter (which that CI job passes) makes CHECK_THROWS_AS() a no-op that never even evaluates its expression -- so patch_inplace() is never called and the follow-up assertions fail against the untouched original document. Guard the whole SECTION with #if !defined(JSON_NOEXCEPTION), following the same convention already used elsewhere in the test suite (e.g. unit-class_parser.cpp) for exception-dependent tests. Signed-off-by: Niels Lohmann * Fix MSVC C2220 in the std::u8string conversion test MSVC's C5321 ("nonstandard extension used: encoding '\xNN' as a multi-byte utf-8 character") is promoted to a hard error by our MSVC CI configs. It fires because the test composed a non-ASCII UTF-8 sequence inside a u8"" literal using raw \x byte escapes; MSVC treats that as nonstandard and suggests using \u universal-character-names instead, which every compiler agrees on and which compiles down to the exact same encoded bytes. Signed-off-by: Niels Lohmann * Guard the JSON_THROW_USER test against JSON_NOEXCEPTION and GCC's -Wunused-result Two independent CI configurations failed to build/run this new test: - ci_test_noexceptions runs the whole suite with -DJSON_NOEXCEPTION and doctest's "--no-throw" filter, which compiles CHECK_THROWS_AS() down to a no-op that never even invokes the guarded expression. Since this test's whole point is to observe json_throw_user_call_count after json::parse()/at() actually throw, it can't be meaningfully run under that filter (our JSON_THROW_USER override still throws real exceptions regardless of JSON_NOEXCEPTION, but the assertion never gets a chance to run). Guard the TEST_CASE with #if !defined(JSON_NOEXCEPTION), mirroring the existing precedent in unit-json_patch.cpp. - ci_test_gcc and ci_test_standards_gcc(11) failed with -Werror=unused-result on the discarded json::parse() return value. json::parse() is marked warn_unused_result, and unlike a real [[nodiscard]] attribute, GCC does not consider that satisfied by doctest's (void)-cast around the expression in C++11 mode. Assign the result to a discarded local instead, matching the established `json _ = json::parse(...)` idiom already used throughout unit-class_parser.cpp. Signed-off-by: Niels Lohmann * Suppress a clang-tidy false positive on an intentional defensive copy performance-unnecessary-copy-initialization suggests copy_for_patch could be a reference since it's never modified -- but the copy is the point: it guards against a hypothetical regression where patch() mutates its receiver, which a reference could never catch (the follow-up assertion would just compare `original` to itself). Signed-off-by: Niels Lohmann * Fix clang-tidy findings in the JSON_NO_IO/JSON_THROW_USER test - bugprone-macro-parentheses: wrap the JSON_THROW_USER macro argument in parentheses at the throw site. - modernize-raw-string-literal: switch two escaped JSON string literals to raw string literals. Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- .github/workflows/ubuntu.yml | 2 +- cmake/ci.cmake | 34 +++++++ tests/src/skip_library_version_check.cpp | 61 ++++++++++++ .../src/unit-byte_container_with_subtype.cpp | 33 +++++++ tests/src/unit-conversions.cpp | 34 +++++++ tests/src/unit-json_patch.cpp | 96 +++++++++++++++++++ .../src/unit-no-mem-leak-on-adl-serialize.cpp | 2 +- tests/src/unit-no_io_and_user_exceptions.cpp | 91 ++++++++++++++++++ tests/src/unit-regression3.cpp | 10 ++ tests/src/unit-std-format.cpp | 17 ++++ 10 files changed, 378 insertions(+), 2 deletions(-) create mode 100644 tests/src/skip_library_version_check.cpp create mode 100644 tests/src/unit-no_io_and_user_exceptions.cpp diff --git a/.github/workflows/ubuntu.yml b/.github/workflows/ubuntu.yml index 69a3cbc45..7ac4cfcb0 100644 --- a/.github/workflows/ubuntu.yml +++ b/.github/workflows/ubuntu.yml @@ -100,7 +100,7 @@ jobs: container: ubuntu:focal strategy: matrix: - target: [ci_cmake_flags, ci_test_diagnostics, ci_test_diagnostic_positions, ci_test_noexceptions, ci_test_noimplicitconversions, ci_test_legacycomparison, ci_test_noglobaludls, ci_test_simdutf] + target: [ci_cmake_flags, ci_test_diagnostics, ci_test_diagnostic_positions, ci_test_noexceptions, ci_test_noimplicitconversions, ci_test_legacycomparison, ci_test_noglobaludls, ci_test_disableenumserialization, ci_test_skiplibraryversioncheck, ci_test_simdutf] steps: - name: Install build-essential run: apt-get update ; apt-get install -y build-essential unzip wget git libssl-dev diff --git a/cmake/ci.cmake b/cmake/ci.cmake index 18fef2075..7d085fb2b 100644 --- a/cmake/ci.cmake +++ b/cmake/ci.cmake @@ -260,6 +260,40 @@ add_custom_target(ci_test_noglobaludls COMMENT "Compile and test with global UDLs disabled" ) +############################################################################### +# Disable enum serialization. +############################################################################### + +add_custom_target(ci_test_disableenumserialization + COMMAND ${CMAKE_COMMAND} + -DCMAKE_BUILD_TYPE=Debug -GNinja + -DJSON_BuildTests=ON -DJSON_FastTests=ON -DJSON_DisableEnumSerialization=ON + -S${PROJECT_SOURCE_DIR} -B${PROJECT_BINARY_DIR}/build_disableenumserialization + COMMAND ${CMAKE_COMMAND} --build ${PROJECT_BINARY_DIR}/build_disableenumserialization + COMMAND cd ${PROJECT_BINARY_DIR}/build_disableenumserialization && ${CMAKE_CTEST_COMMAND} --parallel ${N} --output-on-failure + COMMENT "Compile and test with enum serialization disabled" +) + +############################################################################### +# Skip the multiple-inclusion library version check. +############################################################################### + +# tests/src/skip_library_version_check.cpp deliberately simulates a scenario +# (mixing two differently-versioned inclusions of the library in one +# translation unit) that unavoidably triggers the compiler's own "macro +# redefined" warning, so -- unlike the ci_test_* targets above -- it is +# compiled directly here, with a modest warning set, instead of being folded +# into the library's own -Weverything/-Werror unit test matrix. +add_custom_target(ci_test_skiplibraryversioncheck + COMMAND ${CMAKE_COMMAND} -E make_directory ${PROJECT_BINARY_DIR}/skip_library_version_check + COMMAND ${CMAKE_CXX_COMPILER} -std=c++11 -Wall -Wextra + -I${PROJECT_SOURCE_DIR}/include + ${PROJECT_SOURCE_DIR}/tests/src/skip_library_version_check.cpp + -o ${PROJECT_BINARY_DIR}/skip_library_version_check/skip_library_version_check + COMMAND ${PROJECT_BINARY_DIR}/skip_library_version_check/skip_library_version_check + COMMENT "Compile and run a translation unit simulating a mismatched library version, with JSON_SKIP_LIBRARY_VERSION_CHECK defined" +) + ############################################################################### # Coverage. ############################################################################### diff --git a/tests/src/skip_library_version_check.cpp b/tests/src/skip_library_version_check.cpp new file mode 100644 index 000000000..ddaa4415c --- /dev/null +++ b/tests/src/skip_library_version_check.cpp @@ -0,0 +1,61 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +// Standalone compile-and-run check for the JSON_SKIP_LIBRARY_VERSION_CHECK +// configuration macro, which (per #5423) was never exercised anywhere in the +// test matrix. +// +// include/nlohmann/detail/abi_macros.hpp normally emits a #warning if +// NLOHMANN_JSON_VERSION_MAJOR/MINOR/PATCH are already defined (as they would +// be by an earlier inclusion of a different version of the library) with +// values that mismatch the version about to be defined -- unless +// JSON_SKIP_LIBRARY_VERSION_CHECK is defined, in which case the check (and +// that #warning) is skipped. +// +// This file deliberately is not named tests/src/unit-*.cpp: it is compiled +// directly (with a modest, non-strict warning set) by the dedicated +// ci_test_skiplibraryversioncheck target in cmake/ci.cmake, rather than being +// folded into the library's own -Weverything/-Werror unit test matrix. That +// is because the scenario simulated here -- mixing two different, already +// differently-versioned inclusions of the library in one translation unit -- +// unavoidably also triggers the *compiler's own* "macro redefined" warning, +// independent of (and unaffected by) JSON_SKIP_LIBRARY_VERSION_CHECK, which +// only ever silences the library's own #warning. Building this file under +// -Weverything -Werror would therefore fail for a reason unrelated to the +// macro under test. +#define NLOHMANN_JSON_VERSION_MAJOR 0 +#define NLOHMANN_JSON_VERSION_MINOR 0 +#define NLOHMANN_JSON_VERSION_PATCH 0 + +#define JSON_SKIP_LIBRARY_VERSION_CHECK 1 + +#include + +int main() +{ + // reaching this point at all already proves that the mismatched, + // pre-defined version macros above did not stop compilation -- which is + // exactly what JSON_SKIP_LIBRARY_VERSION_CHECK is for. The library must + // also still be fully usable. + const nlohmann::json j = {{"a", 1}, {"b", {1, 2, 3}}}; + if (j.dump() != "{\"a\":1,\"b\":[1,2,3]}") + { + return 1; + } + + // include/nlohmann/detail/abi_macros.hpp unconditionally (re)defines the + // version macros to the library's real, current version right after the + // (here, skipped) mismatch check, regardless of the deliberately wrong + // stand-in values defined above. + if (NLOHMANN_JSON_VERSION_MAJOR == 0 && NLOHMANN_JSON_VERSION_MINOR == 0 && NLOHMANN_JSON_VERSION_PATCH == 0) + { + return 1; + } + + return 0; +} diff --git a/tests/src/unit-byte_container_with_subtype.cpp b/tests/src/unit-byte_container_with_subtype.cpp index 3983ba95f..2e448ac7d 100644 --- a/tests/src/unit-byte_container_with_subtype.cpp +++ b/tests/src/unit-byte_container_with_subtype.cpp @@ -42,6 +42,39 @@ TEST_CASE("byte_container_with_subtype") CHECK(container.subtype() == static_cast(-1)); } + SECTION("move semantics") + { + // the rvalue-reference constructor (without a subtype) must actually move + // the passed-in container rather than copy it; comparing the buffer address + // before and after is a stronger check than just observing the source is + // empty afterward, since a copy-then-clear could also leave it empty + { + std::vector bytes = {{0xCA, 0xFE, 0xBA, 0xBE}}; + const auto* const data_ptr = bytes.data(); + + nlohmann::byte_container_with_subtype> container(std::move(bytes)); + + CHECK(container.size() == 4); + CHECK(container.data() == data_ptr); + CHECK(!container.has_subtype()); + CHECK(bytes.empty()); // NOLINT(bugprone-use-after-move,clang-analyzer-cplusplus.Move,hicpp-invalid-access-moved) + } + + // same check for the rvalue-reference constructor that also takes a subtype + { + std::vector bytes = {{0xCA, 0xFE, 0xBA, 0xBE}}; + const auto* const data_ptr = bytes.data(); + + nlohmann::byte_container_with_subtype> container(std::move(bytes), 42); + + CHECK(container.size() == 4); + CHECK(container.data() == data_ptr); + CHECK(container.has_subtype()); + CHECK(container.subtype() == 42); + CHECK(bytes.empty()); // NOLINT(bugprone-use-after-move,clang-analyzer-cplusplus.Move,hicpp-invalid-access-moved) + } + } + SECTION("comparisons") { std::vector const bytes = {{0xCA, 0xFE, 0xBA, 0xBE}}; diff --git a/tests/src/unit-conversions.cpp b/tests/src/unit-conversions.cpp index 4975854c0..90d972f71 100644 --- a/tests/src/unit-conversions.cpp +++ b/tests/src/unit-conversions.cpp @@ -1792,6 +1792,40 @@ TEST_CASE("std::filesystem::path") } #endif +// the ADL to_json overload for std::u8string only exists under the same guard +// as std::filesystem::path support (it is otherwise only reached indirectly, +// via std::filesystem::path::u8string()) -- mirror both #if conditions from +// include/nlohmann/detail/conversions/to_json.hpp exactly +#if JSON_HAS_FILESYSTEM || JSON_HAS_EXPERIMENTAL_FILESYSTEM +#if defined(__cpp_lib_char8_t) +TEST_CASE("std::u8string") +{ + SECTION("ascii") + { + const std::u8string s = u8"Path"; + json const j = s; + + CHECK(j.template get() == "Path"); + } + + SECTION("utf-8") + { + // use \u universal-character-names (rather than raw \x byte escapes + // or literal non-ASCII source bytes) to compose the multi-byte UTF-8 + // encoding -- MSVC treats \x escapes used that way inside a u8 + // literal as a nonstandard extension (warning C5321), which some of + // our CI configs promote to an error; \u is portable and produces + // the exact same encoded bytes without depending on the source + // file's encoding + const std::u8string s = u8"P\u011B\u0161ina"; + json const j = s; + + CHECK(j.template get() == "P\xc4\x9b\xc5\xa1ina"); + } +} +#endif +#endif + TEST_CASE("std::optional") { SECTION("null") diff --git a/tests/src/unit-json_patch.cpp b/tests/src/unit-json_patch.cpp index 257e455aa..7731c7d92 100644 --- a/tests/src/unit-json_patch.cpp +++ b/tests/src/unit-json_patch.cpp @@ -672,6 +672,102 @@ TEST_CASE("JSON patch") } } + SECTION("patch_inplace") + { + SECTION("happy path: patch_inplace mirrors patch() on success") + { + // mirrors "A.5. Replacing a Value" above, but applies the patch with + // patch_inplace() to a mutable copy instead of using patch()'s + // returned copy + json doc = R"( + { + "baz": "qux", + "foo": "bar" + } + )"_json; + + json const patch = R"( + [ + { "op": "replace", "path": "/baz", "value": "boo" } + ] + )"_json; + + json const expected = R"( + { + "baz": "boo", + "foo": "bar" + } + )"_json; + + doc.patch_inplace(patch); + CHECK(doc == expected); + } + + // this test relies on the "test" operation actually throwing so the + // partial-application state can be observed right after the throw + // point; under JSON_NOEXCEPTION, JSON_THROW() calls std::abort() + // instead (there is no C++ exception to throw), and doctest's + // CHECK_THROWS_AS() is compiled out to a no-op that never even + // invokes the given expression (see doctest's "--no-throw" test + // filter, which ci_test_noexceptions passes) -- so patch()/ + // patch_inplace() would never be called at all and the follow-up + // state assertions below would fail against the untouched original +#if !defined(JSON_NOEXCEPTION) + SECTION("distinguishing contract vs patch(): partial application on failure") + { + // Unlike patch(), which is all-or-nothing because it applies the + // patch to an internal copy that is simply discarded when an + // exception is thrown (leaving the original untouched no matter + // what), patch_inplace() mutates the document it is called on + // directly and immediately, operation by operation. So if a JSON + // Patch fails partway through, whatever operations already + // succeeded remain applied -- the document is left in a partially + // patched state. This is empirically verified current behavior, + // not just documented intent, and is pinned here as such. + json const original = R"( + { + "baz": "qux", + "foo": "bar" + } + )"_json; + + // the first operation ("replace") succeeds; the second ("test") + // fails because the value at "/baz" no longer (and never did) + // equal "not boo" + json const patch = R"( + [ + { "op": "replace", "path": "/baz", "value": "boo" }, + { "op": "test", "path": "/baz", "value": "not boo" } + ] + )"_json; + + // patch() never modifies the object it is called on -- it always + // operates on (and returns) a separate copy, so the original is + // left completely untouched, regardless of success or failure. + // copy_for_patch is intentionally a real copy, not a reference + // to `original`: the whole point of this check is to catch a + // hypothetical future regression where patch() *does* mutate its + // receiver. Using a reference here would make the assertion + // below compare `original` to itself -- trivially true even if + // such a bug existed -- which is exactly what a static analyzer + // can't see when it suggests "this copy is never modified, use + // a reference instead". + json copy_for_patch = original; // NOLINT(performance-unnecessary-copy-initialization) + CHECK_THROWS_AS(copy_for_patch.patch(patch), json::other_error&); + CHECK(copy_for_patch == original); + + // patch_inplace(), in contrast, already applied the successful + // "replace" operation to the document before the "test" operation + // threw -- that change is not rolled back + json doc = original; + CHECK_THROWS_AS(doc.patch_inplace(patch), json::other_error&); + CHECK(doc != original); + CHECK(doc.at("baz") == "boo"); + CHECK(doc.at("foo") == "bar"); + } +#endif // !defined(JSON_NOEXCEPTION) + } + SECTION("errors") { SECTION("unknown operation") diff --git a/tests/src/unit-no-mem-leak-on-adl-serialize.cpp b/tests/src/unit-no-mem-leak-on-adl-serialize.cpp index 469fc2c75..cfbdff008 100644 --- a/tests/src/unit-no-mem-leak-on-adl-serialize.cpp +++ b/tests/src/unit-no-mem-leak-on-adl-serialize.cpp @@ -70,7 +70,7 @@ TEST_CASE("check_for_mem_leak_on_adl_to_json-2") } } -TEST_CASE("check_for_mem_leak_on_adl_to_json-2") +TEST_CASE("check_for_mem_leak_on_adl_to_json-3") { try { diff --git a/tests/src/unit-no_io_and_user_exceptions.cpp b/tests/src/unit-no_io_and_user_exceptions.cpp new file mode 100644 index 000000000..667d114e7 --- /dev/null +++ b/tests/src/unit-no_io_and_user_exceptions.cpp @@ -0,0 +1,91 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +// This translation unit is a dedicated, small compile-and-run check for two +// configuration macros that (per #5423) were never exercised anywhere in the +// test matrix: +// - JSON_NO_IO, which removes the library's / support +// (operator<<, operator>>, and the stream-based overloads of dump()/parse()) +// - the JSON_THROW_USER / JSON_TRY_USER / JSON_CATCH_USER trio, which lets a +// user replace the library's internal exception handling +// +// Both macros are about excluding/replacing a facility the library would +// otherwise pull in on its own, and defining one has no bearing on the other, +// so -- to keep the test matrix small -- they are exercised together in a +// single dedicated file instead of two. +// +// JSON_NO_IO requires this file itself to never rely on /; +// only string-based parsing/dumping is used below. +#define JSON_NO_IO 1 + +// The user-supplied exception macros below are a *conforming* replacement: +// they simply forward to the real throw/try/catch keywords (via a counter so +// the test can assert each macro was actually invoked, not just defined), so +// every exception-related behavior the library relies on internally -- +// including rethrowing std::out_of_range as json::out_of_range in at() -- +// keeps working exactly as it would with the library's own default macros. +static int json_throw_user_call_count = 0; // NOLINT(cppcoreguidelines-avoid-non-const-global-variables) + +#define JSON_THROW_USER(exception) do { ++json_throw_user_call_count; throw (exception); } while (false) // NOLINT(cppcoreguidelines-macro-usage) +#define JSON_TRY_USER try // NOLINT(cppcoreguidelines-macro-usage) +#define JSON_CATCH_USER(exception) catch (exception) // NOLINT(cppcoreguidelines-macro-usage) + +#include "doctest_compatibility.h" + +#include +using json = nlohmann::json; + +TEST_CASE("JSON_NO_IO") +{ + // everything that does not touch / must keep working: + // parsing from and dumping to std::string + const json j = json::parse(R"({"a":[1,2,3],"b":true})"); + CHECK(j.dump() == R"({"a":[1,2,3],"b":true})"); + CHECK(j.at("a").size() == 3); + CHECK(j.at("b").get() == true); +} + +// this test relies on CHECK_THROWS_AS() actually invoking the guarded +// expression so json_throw_user_call_count gets bumped and can be observed +// afterwards; doctest's "--no-throw" test filter (which ci_test_noexceptions +// passes, together with a global -DJSON_NOEXCEPTION added to CMAKE_CXX_FLAGS +// for every translation unit in that build, this file included) compiles +// CHECK_THROWS_AS() out to a no-op that never even invokes the given +// expression -- so json::parse()/at() below would never be called at all and +// the call-count assertions would fail even though our JSON_THROW_USER +// override (which always really throws, regardless of JSON_NOEXCEPTION) would +// have worked fine on its own +#if !defined(JSON_NOEXCEPTION) +TEST_CASE("JSON_THROW_USER, JSON_TRY_USER, JSON_CATCH_USER") +{ + json_throw_user_call_count = 0; + + // json::parse() is [[nodiscard]] (JSON_HEDLEY_WARN_UNUSED_RESULT); under + // GCC in C++11 mode that expands to __attribute__((warn_unused_result)), + // which -- unlike a [[nodiscard]] attribute proper -- GCC does not + // consider satisfied by doctest's CHECK_THROWS_AS() wrapping the + // expression in a (void) cast, so the discarded return value would still + // be flagged under -Werror=unused-result; assign it to discard it instead, + // matching the established `json _ = json::parse(...)` pattern used + // elsewhere in the test suite (see unit-class_parser.cpp) + json _; // NOLINT(readability-identifier-naming) + + // a parse error goes through JSON_THROW directly, i.e., through our + // JSON_THROW_USER override + CHECK_THROWS_AS(_ = json::parse("this is not JSON"), json::parse_error&); + CHECK(json_throw_user_call_count > 0); + + // at() on an out-of-range array index internally catches std::out_of_range + // (JSON_TRY_USER/JSON_CATCH_USER) and rethrows it as json::out_of_range + // (JSON_THROW_USER again), so this exercises all three macros together + const int count_before = json_throw_user_call_count; + const json arr = json::array({1, 2, 3}); + CHECK_THROWS_AS(arr.at(10), json::out_of_range&); + CHECK(json_throw_user_call_count > count_before); +} +#endif diff --git a/tests/src/unit-regression3.cpp b/tests/src/unit-regression3.cpp index 11c6a7da8..a5be9ec4b 100644 --- a/tests/src/unit-regression3.cpp +++ b/tests/src/unit-regression3.cpp @@ -18,6 +18,14 @@ // for some reason including this after the json header leads to linker errors with VS 2017... #include +// skip tests if JSON_DisableEnumSerialization=ON (#4384): std::byte is a +// scoped enum, so get() (needed below to get>() +// from a plain JSON array, not just from an already-binary value) relies on +// enum serialization being enabled +#if defined(JSON_DISABLE_ENUM_SERIALIZATION) && (JSON_DISABLE_ENUM_SERIALIZATION == 1) + #define SKIP_TESTS_FOR_ENUM_SERIALIZATION +#endif + #define JSON_TESTS_PRIVATE #include using json = nlohmann::json; @@ -466,6 +474,7 @@ TEST_CASE("regression tests 3") CHECK((decoded == json_4804::array())); } +#ifndef SKIP_TESTS_FOR_ENUM_SERIALIZATION SECTION("discussion #4209 - custom BinaryType direct assignment and round-tripping") { // Test that assigning a custom BinaryType directly creates a binary value, not an array @@ -499,6 +508,7 @@ TEST_CASE("regression tests 3") CHECK(extracted[1] == std::byte{2}); CHECK(extracted[2] == std::byte{3}); } +#endif SECTION("issue #5046 - implicit conversion of return json to std::optional no longer implicit") { diff --git a/tests/src/unit-std-format.cpp b/tests/src/unit-std-format.cpp index d18ccbd32..58cbbf5cc 100644 --- a/tests/src/unit-std-format.cpp +++ b/tests/src/unit-std-format.cpp @@ -53,6 +53,23 @@ TEST_CASE("std::formatter") CHECK(std::format("{:2}", j) == j.dump(2)); CHECK(std::format("{:#2}", j) == j.dump(2)); CHECK(std::format("{:8}", j) == j.dump(8)); + // multi-digit widths must accumulate every digit, not just the first + CHECK(std::format("{:12}", j) == j.dump(12)); + CHECK(std::format("{:#12}", j) == j.dump(12)); + CHECK(std::format("{:10}", j) == j.dump(10)); + } + + SECTION("bare alignment with no fill character defaults to a space indent character") + { + const json j = {{"foo", 1}, {"bar", {1, 2, 3}}}; + // without a preceding fill character, the alignment character itself must not + // be mistaken for the indent character -- the default space is kept + CHECK(std::format("{:<}", j) == j.dump()); + CHECK(std::format("{:>}", j) == j.dump()); + CHECK(std::format("{:^}", j) == j.dump()); + CHECK(std::format("{:<3}", j) == j.dump(3, ' ')); + CHECK(std::format("{:>3}", j) == j.dump(3, ' ')); + CHECK(std::format("{:^3}", j) == j.dump(3, ' ')); } SECTION("fill-and-align sets the indent character, like dump(indent, indent_char)") From 29ba5973b697c014b64a2ec051f1d75e70f0f6ab Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Wed, 16 Sep 2026 20:16:40 +0200 Subject: [PATCH 07/64] Add missing diagnostic-positions test coverage (lifetime, input adapters, SAX) (#5482) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * Add missing diagnostic-positions test coverage (lifetime, input adapters, SAX) Building on the merged unit-class_parser.cpp from #5417, add characterization tests (regression protection for existing behavior, not a behavior change) for JSON_DIAGNOSTIC_POSITIONS: - value lifetime: copy ctor copies positions recursively, move ctor resets the moved-from value to npos, and mutating a parsed document (operator[], push_back, erase) leaves the parent's stale span and siblings' positions untouched while new values get npos. - input adapters: wide-string input positions count transcoded UTF-8 bytes (not wide characters), BOM-prefixed input's start_pos() reflects the skipped 3-byte BOM, istringstream/ifstream/iterator-pair inputs report consistent (non-npos) positions, and binary formats (CBOR, MessagePack, UBJSON, BSON) always report npos. - a user-constructed json_sax_dom_parser with no lexer (as used when driving json::sax_parse() directly) reports npos for every value, since it has no m_lexer_ref to source positions from. While characterizing swap(), found that basic_json::swap() (and the friend swap() that forwards to it) does not swap start_position/end_position, unlike copy-assignment's operator=(basic_json), which does as part of its copy-and-swap implementation. This looks like a real inconsistency/bug, but per the scope of this test-only change it is only pinned (not fixed) here; see the comment at the "swap() does NOT exchange positions" section. Fixes #5420 Signed-off-by: Niels Lohmann * Fix MSVC source-encoding portability in the wide-string position test Use é escapes instead of a literal UTF-8-encoded 'é' inside the L"" literal, so the wide string's content does not depend on the compiler's assumed source character set (MSVC without /utf-8 decodes raw non-ASCII source bytes using the system code page rather than as UTF-8, which was producing a wstring of unexpected length/content and failing the ws.size()/end_pos() assertions on Windows CI). Also reworded a comment that unintentionally embedded the literal substring "TODO check", which clang-tidy's google-readability-todo check flags regardless of quoting context. Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- tests/src/unit-class_parser.cpp | 320 ++++++++++++++++++++++++++++++++ 1 file changed, 320 insertions(+) diff --git a/tests/src/unit-class_parser.cpp b/tests/src/unit-class_parser.cpp index af76e93cc..5b4af321c 100644 --- a/tests/src/unit-class_parser.cpp +++ b/tests/src/unit-class_parser.cpp @@ -17,6 +17,8 @@ using nlohmann::json; #include #include +#include +#include #include #include #include @@ -2445,3 +2447,321 @@ TEST_CASE("last-read diagnostics are identical across input adapters") } } #endif // !defined(JSON_NOEXCEPTION) + +// this test characterizes the current (documented-by-example, not otherwise +// specified) behavior of JSON_DIAGNOSTIC_POSITIONS positions with respect to +// value lifetime (copy/move/swap/mutation), the various input adapters, and +// user-driven SAX usage. It is regression protection, not a behavior +// specification: if any of these checks fail after a change to json.hpp, +// that change deliberately altered observable behavior and the test (and +// this comment) should be updated accordingly, rather than "fixed" blindly. +#if JSON_DIAGNOSTIC_POSITIONS +TEST_CASE("diagnostic positions: value lifetime, input adapters, and SAX") +{ + SECTION("value lifetime") + { + SECTION("copy constructor copies positions, recursively") + { + // basic_json(const basic_json&) (json.hpp, around line 1192) copies + // start_position/end_position for the value itself; nested values + // are copied via their own copy constructor (through the copied + // object/array container), so positions are preserved throughout + // the whole tree. + const std::string s = R"({"a":1,"b":[1,2,3]})"; + const json a = json::parse(s); + const json b = a; // NOLINT(performance-unnecessary-copy-initialization) + + CHECK(b.start_pos() == a.start_pos()); + CHECK(b.end_pos() == a.end_pos()); + CHECK(b["b"].start_pos() == a["b"].start_pos()); + CHECK(b["b"].end_pos() == a["b"].end_pos()); + CHECK(b["b"][0].start_pos() == a["b"][0].start_pos()); + CHECK(b["b"][0].end_pos() == a["b"][0].end_pos()); + + // sanity: the positions are meaningful (not all npos) + CHECK(b.start_pos() == 0); + CHECK(b.end_pos() == s.size()); + } + + SECTION("move constructor resets the moved-from value to npos") + { + // basic_json(basic_json&&) (json.hpp, around line 1265) copies + // other's start_position/end_position into *this and then resets + // other's to npos (see the cppcheck-suppress[accessForwarded] + // annotation there, which flags this reset as worth a second + // look). Only the top-level moved-from value is affected; its + // (moved-away) children are gone along with it. + const std::string s = R"({"a":1,"b":[1,2,3]})"; + json a = json::parse(s); + const auto a_start = a.start_pos(); + const auto a_end = a.end_pos(); + const auto nested_start = a["b"].start_pos(); + const auto nested_end = a["b"].end_pos(); + + const json b(std::move(a)); + + // the destination retains the original positions, recursively + CHECK(b.start_pos() == a_start); + CHECK(b.end_pos() == a_end); + CHECK(b["b"].start_pos() == nested_start); + CHECK(b["b"].end_pos() == nested_end); + + // the moved-from value is reset to a null and reports npos + CHECK(a.is_null()); // NOLINT(bugprone-use-after-move,clang-analyzer-cplusplus.Move) + CHECK(a.start_pos() == std::string::npos); // NOLINT(bugprone-use-after-move,clang-analyzer-cplusplus.Move) + CHECK(a.end_pos() == std::string::npos); // NOLINT(bugprone-use-after-move,clang-analyzer-cplusplus.Move) + } + + SECTION("swap() does NOT exchange positions (likely a real bug, see below)") + { + // NOTE (characterizing, not fixing, for #5420): basic_json::swap() + // (json.hpp, around line 3540, and the friend swap() that forwards + // to it) swaps m_data.m_type and m_data.m_value but -- unlike + // copy-assignment's operator=(basic_json) (json.hpp, around line + // 1291), which swaps start_position/end_position as part of its + // copy-and-swap implementation -- it never touches + // start_position/end_position. So after swap(a, b), the *values* + // of a and b are exchanged, but their *positions* are not: each + // ends up with its own original position describing the other's + // new content. This looks like an oversight/inconsistency rather + // than intended behavior, and is flagged to the maintainer; this + // test only pins the current (surprising) behavior so a fix (or a + // deliberate decision to keep it) shows up here as an intentional + // change rather than a silent regression. + json a = json::parse(R"({"a":1})"); + json b = json::parse(R"([1,2,3,4,5])"); + const auto a_start = a.start_pos(); + const auto a_end = a.end_pos(); + const auto b_start = b.start_pos(); + const auto b_end = b.end_pos(); + // both start at 0 (root values start right away), but their + // lengths (and thus end positions) differ, which is enough to + // tell after the swap whether positions actually moved with + // the values + CHECK(a_end != b_end); + + using std::swap; + swap(a, b); + + // values were exchanged as expected ... + CHECK(a == json::parse(R"([1,2,3,4,5])")); + CHECK(b == json::parse(R"({"a":1})")); + + // ... but positions were NOT: each variable kept its own + // original position, now describing the other's content + CHECK(a.start_pos() == a_start); + CHECK(a.end_pos() == a_end); + CHECK(b.start_pos() == b_start); + CHECK(b.end_pos() == b_end); + } + + SECTION("mutating a parsed document leaves positions of unrelated values untouched") + { + // Positions are recorded once, during parsing, and are not + // recomputed on mutation. As a consequence, after a mutation the + // parent's own recorded span may no longer describe its current + // (serialized) content -- it still describes what was originally + // parsed. This is characterized here as current behavior, not + // asserted to be desirable or specified. + SECTION("operator[] adding a new object key") + { + const std::string s = R"({"a":1})"; + json j = json::parse(s); + const auto root_start = j.start_pos(); + const auto root_end = j.end_pos(); + const auto a_start = j["a"].start_pos(); + const auto a_end = j["a"].end_pos(); + + j["c"] = 42; + + // the newly-added value was never parsed, so it has no position + CHECK(j["c"].start_pos() == std::string::npos); + CHECK(j["c"].end_pos() == std::string::npos); + + // the existing sibling's position is unaffected + CHECK(j["a"].start_pos() == a_start); + CHECK(j["a"].end_pos() == a_end); + + // the parent's own recorded span is left as-is (now stale: + // it still reflects the original, shorter `{"a":1}` string) + CHECK(j.start_pos() == root_start); + CHECK(j.end_pos() == root_end); + } + + SECTION("push_back on a parsed array") + { + const std::string s = R"([1,2,3])"; + json j = json::parse(s); + const auto root_start = j.start_pos(); + const auto root_end = j.end_pos(); + const auto first_start = j[0].start_pos(); + + j.push_back(4); + + CHECK(j.back().start_pos() == std::string::npos); + CHECK(j.back().end_pos() == std::string::npos); + CHECK(j[0].start_pos() == first_start); + CHECK(j.start_pos() == root_start); + CHECK(j.end_pos() == root_end); + } + + SECTION("erase on a parsed array shifts elements but keeps their own positions") + { + const std::string s = R"([1,2,3])"; + json j = json::parse(s); + const auto second_start = j[1].start_pos(); + const auto third_start = j[2].start_pos(); + const auto root_start = j.start_pos(); + const auto root_end = j.end_pos(); + + j.erase(0); + + // remaining elements moved down an index, but each one still + // reports the position it had *before* the erase (i.e. its + // position in the original source string, not a + // recalculated one) + CHECK(j[0].start_pos() == second_start); + CHECK(j[1].start_pos() == third_start); + + // the parent's own recorded span is again left as-is + CHECK(j.start_pos() == root_start); + CHECK(j.end_pos() == root_end); + } + } + } + + SECTION("input adapters") + { + SECTION("wide string input: positions count transcoded UTF-8 bytes, not wide characters") + { + // 'é' (U+00E9) is a single code unit in a wchar_t/UTF-16 string, but + // transcodes to 2 bytes in UTF-8; the lexer only ever sees the + // transcoded UTF-8 byte stream, so reported positions are byte + // offsets into that UTF-8 stream, not indices into the original + // std::wstring. + // é (rather than a literal 'é' byte sequence in this source + // file) so the wide-string literal's meaning does not depend on + // the compiler's assumed source character set (MSVC, without + // /utf-8, would otherwise decode the raw UTF-8 bytes using the + // system code page instead of as UTF-8) + const std::wstring ws = L"{\"a\":\"\u00e9\u00e9\"}"; + CHECK(ws.size() == 10); // 10 wide characters + + const json j = json::parse(ws); + CHECK(j.start_pos() == 0); + // the transcoded UTF-8 form is 2 bytes longer than the wide string, + // because each of the two 'é' characters becomes 2 UTF-8 bytes + CHECK(j.end_pos() == 12); + CHECK(j.end_pos() != ws.size()); + + const json& a = j["a"]; + CHECK(a.start_pos() == 5); + CHECK(a.end_pos() == 11); + } + + SECTION("BOM-prefixed input: start_pos() reflects the skipped 3-byte BOM") + { + const std::string s = "\xEF\xBB\xBF{\"a\":1}"; + const json j = json::parse(s); + + // the lexer silently skips the BOM before parsing the value, so + // the root value's recorded span starts right after it + CHECK(j.start_pos() == 3); + CHECK(j.end_pos() == s.size()); + } + + SECTION("std::istringstream: positions are consistent, not npos") + { + const std::string s = R"({"a":1,"b":2})"; + std::istringstream ss(s); + const json j = json::parse(ss); + + CHECK(j.start_pos() == 0); + CHECK(j.end_pos() == s.size()); + CHECK(j["a"].start_pos() == 5); + } + + SECTION("std::ifstream: positions are consistent, not npos") + { + const std::string s = R"({"a":1,"b":2})"; + { + std::ofstream file("unit-class_parser_diagnostic_positions.tmp"); + file << s; + } + + { + std::ifstream f("unit-class_parser_diagnostic_positions.tmp"); + const json j = json::parse(f); + + CHECK(j.start_pos() == 0); + CHECK(j.end_pos() == s.size()); + CHECK(j["a"].start_pos() == 5); + } + + static_cast(std::remove("unit-class_parser_diagnostic_positions.tmp")); + } + + SECTION("iterator-pair input: positions are consistent, not npos") + { + const std::string s = R"({"a":1,"b":2})"; + const json j = json::parse(s.begin(), s.end()); + + CHECK(j.start_pos() == 0); + CHECK(j.end_pos() == s.size()); + CHECK(j["a"].start_pos() == 5); + } + + SECTION("binary formats have no text positions") + { + // binary formats (CBOR, MessagePack, UBJSON, BSON, BJData) are + // parsed via detail::binary_reader, which never sets + // start_position/end_position on the values it produces (they + // have no notion of a text offset), so every value's position + // stays at its default of npos. + const json src = json::parse(R"({"a":1,"b":[1,2]})"); + + const json from_cbor = json::from_cbor(json::to_cbor(src)); + CHECK(from_cbor.start_pos() == std::string::npos); + CHECK(from_cbor.end_pos() == std::string::npos); + CHECK(from_cbor["a"].start_pos() == std::string::npos); + CHECK(from_cbor["b"][0].start_pos() == std::string::npos); + + const json from_msgpack = json::from_msgpack(json::to_msgpack(src)); + CHECK(from_msgpack.start_pos() == std::string::npos); + CHECK(from_msgpack.end_pos() == std::string::npos); + + const json from_ubjson = json::from_ubjson(json::to_ubjson(src)); + CHECK(from_ubjson.start_pos() == std::string::npos); + CHECK(from_ubjson.end_pos() == std::string::npos); + + const json from_bson_val = json::from_bson(json::to_bson(src)); + CHECK(from_bson_val.start_pos() == std::string::npos); + CHECK(from_bson_val.end_pos() == std::string::npos); + } + } + + SECTION("user-driven SAX consumers with no lexer report npos") + { + // json::parse() internally wires up its json_sax_dom_parser with a + // pointer to its own lexer (see parser.hpp), which is how positions + // get set at all. A user who constructs a json_sax_dom_parser + // directly (e.g. to drive it via json::sax_parse()) and does not + // supply a lexer pointer gets a consumer with m_lexer_ref == nullptr; + // every "if (m_lexer_ref)" guard in json_sax.hpp is then skipped, so + // every value it produces keeps its default, unset position (npos). + // This was previously true but silently unasserted (operator== + // ignores positions), see #5420. + json result; + nlohmann::detail::json_sax_dom_parser sdp(result); + const std::string s = R"({"a":1,"b":[1,2,3]})"; + CHECK(json::sax_parse(s, &sdp)); + + CHECK(result.start_pos() == std::string::npos); + CHECK(result.end_pos() == std::string::npos); + CHECK(result["a"].start_pos() == std::string::npos); + CHECK(result["a"].end_pos() == std::string::npos); + CHECK(result["b"][0].start_pos() == std::string::npos); + CHECK(result["b"][0].end_pos() == std::string::npos); + } +} +#endif From e8e1ba0db9ed3d760da1e9d55ce3dc4f44cc597e Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Wed, 16 Sep 2026 20:16:43 +0200 Subject: [PATCH 08/64] Fix dead ill-formed-fourth-byte UTF-8 test sections (byte3/byte4 typo) (#5499) * Fix dead ill-formed-fourth-byte UTF-8 test sections (byte3/byte4 typo) The "ill-formed: wrong fourth byte" SECTIONs in unit-unicode3.cpp, unit-unicode4.cpp, and unit-unicode5.cpp guarded their loop with a check on byte3 instead of byte4. Since the enclosing loop already restricts byte3 to its valid range, the guard was always true and the section's "continue" fired unconditionally, so check_utf8string()/check_utf8dump() were never actually invoked for a malformed fourth byte. Fixing the guard naively (byte3 -> byte4) would also have swept the full byte2 x byte3 combinatorics for every byte4 value, adding millions of redundant iterations: the lexer validates continuation bytes strictly in sequence with early exit (see next_byte_in_range() in lexer.hpp), so once byte2/byte3 are within their valid range, the byte4 outcome does not depend on which valid byte2/byte3 values were chosen. Instead, byte2 and byte3 are now held to a small hedge of representative valid prefixes (range corners plus a midpoint) while byte4 is still swept exhaustively over its full 0x00-0xFF range, since that is the actual property under test. Also fixed the garbled "skip fourth second byte" comment in unit-unicode3.cpp. Verified offline: before the fix, the "wrong fourth byte" subcase executes 0 assertions in all three files (proving it was dead code); after the fix, it executes 11520 (unicode3), 34560 (unicode4), and 11520 (unicode5) assertions, and a deliberately reintroduced bug in the lexer's byte4 range check causes it to fail (proving it is now meaningful). Total per-file assertion counts grow by the same small amounts, not by millions, and all other sections in these files still pass unchanged. Fixes #5416 Signed-off-by: Niels Lohmann * Keep full byte2 x byte3 combinatorics in the wrong-fourth-byte sections The maintainer wants exhaustive coverage of every byte combination here rather than the representative-prefix reduction, matching the style of the sibling "wrong second/third byte" sections in the same files. Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- tests/src/unit-unicode3.cpp | 4 ++-- tests/src/unit-unicode4.cpp | 2 +- tests/src/unit-unicode5.cpp | 2 +- 3 files changed, 4 insertions(+), 4 deletions(-) diff --git a/tests/src/unit-unicode3.cpp b/tests/src/unit-unicode3.cpp index d5627d8cc..12c12eea4 100644 --- a/tests/src/unit-unicode3.cpp +++ b/tests/src/unit-unicode3.cpp @@ -306,8 +306,8 @@ TEST_CASE("Unicode (3/5)" * doctest::skip()) { for (int byte4 = 0x00; byte4 <= 0xFF; ++byte4) { - // skip fourth second byte - if (0x80 <= byte3 && byte3 <= 0xBF) + // skip correct fourth byte + if (0x80 <= byte4 && byte4 <= 0xBF) { continue; } diff --git a/tests/src/unit-unicode4.cpp b/tests/src/unit-unicode4.cpp index f15a1499f..43cf7095e 100644 --- a/tests/src/unit-unicode4.cpp +++ b/tests/src/unit-unicode4.cpp @@ -307,7 +307,7 @@ TEST_CASE("Unicode (4/5)" * doctest::skip()) for (int byte4 = 0x00; byte4 <= 0xFF; ++byte4) { // skip correct fourth byte - if (0x80 <= byte3 && byte3 <= 0xBF) + if (0x80 <= byte4 && byte4 <= 0xBF) { continue; } diff --git a/tests/src/unit-unicode5.cpp b/tests/src/unit-unicode5.cpp index e35801823..bc0312820 100644 --- a/tests/src/unit-unicode5.cpp +++ b/tests/src/unit-unicode5.cpp @@ -307,7 +307,7 @@ TEST_CASE("Unicode (5/5)" * doctest::skip()) for (int byte4 = 0x00; byte4 <= 0xFF; ++byte4) { // skip correct fourth byte - if (0x80 <= byte3 && byte3 <= 0xBF) + if (0x80 <= byte4 && byte4 <= 0xBF) { continue; } From 502e9d66f64feb378b7a89b17fc4a40d4c63a45b Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Wed, 16 Sep 2026 20:20:21 +0200 Subject: [PATCH 09/64] Fix to_bjdata() emitting the Draft-3-only 'B' marker in default Draft-2 mode (#5479) * Fix to_bjdata() emitting the Draft-3-only 'B' marker in default Draft-2 mode _ArrayType_ = "byte" mapped unconditionally to the BJData type marker 'B', regardless of the requested bjdata_version. 'B' is defined only by BJData Draft 3; with the default version (draft2), this produced a stream that is invalid for Draft 2 and, unlike every other _ArrayType_, round-tripped back as a binary value instead of the original annotated object. Only accept "byte" / emit 'B' when bjdata_version selects Draft 3. Under Draft 2, fall back to the same plain-object encoding used elsewhere in this function for other invalid-annotation cases, so the value round-trips correctly. Fixes #5404. Signed-off-by: Niels Lohmann * Future-proof the Draft-3-only 'B' marker gate @gregmarr pointed out that dtype == 'B' && bjdata_version != draft3 only future-proofs by accident, since bjdata_version_t currently has exactly two values. Compare with < instead, so a later draft that keeps the 'B' marker valid does not need this gate revisited. Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- .../nlohmann/detail/output/binary_writer.hpp | 10 ++++ single_include/nlohmann/json.hpp | 10 ++++ tests/src/unit-bjdata.cpp | 49 ++++++++++++++++++- 3 files changed, 67 insertions(+), 2 deletions(-) diff --git a/include/nlohmann/detail/output/binary_writer.hpp b/include/nlohmann/detail/output/binary_writer.hpp index e9ccd23b5..aaa638801 100644 --- a/include/nlohmann/detail/output/binary_writer.hpp +++ b/include/nlohmann/detail/output/binary_writer.hpp @@ -1707,6 +1707,16 @@ class binary_writer } CharType dtype = it->second; + // the 'B' (byte) marker is only defined from BJData Draft 3 onward; + // emitting it under an earlier draft would produce a stream that an + // earlier-draft reader rejects, so such an object falls back to a + // plain object encoding instead (see the "Binary values" section of + // the BJData documentation) + if (dtype == 'B' && bjdata_version < bjdata_version_t::draft3) + { + return true; + } + key = "_ArraySize_"; // the dimensions are written verbatim as the header length below, so a // value that is not an array cannot produce a valid one: null emits 'Z' diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 35443e141..4253fcdcf 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -20384,6 +20384,16 @@ class binary_writer } CharType dtype = it->second; + // the 'B' (byte) marker is only defined from BJData Draft 3 onward; + // emitting it under an earlier draft would produce a stream that an + // earlier-draft reader rejects, so such an object falls back to a + // plain object encoding instead (see the "Binary values" section of + // the BJData documentation) + if (dtype == 'B' && bjdata_version < bjdata_version_t::draft3) + { + return true; + } + key = "_ArraySize_"; // the dimensions are written verbatim as the header length below, so a // value that is not an array cannot produce a valid one: null emits 'Z' diff --git a/tests/src/unit-bjdata.cpp b/tests/src/unit-bjdata.cpp index 9338e663e..a53bd17ce 100644 --- a/tests/src/unit-bjdata.cpp +++ b/tests/src/unit-bjdata.cpp @@ -2586,7 +2586,12 @@ TEST_CASE("BJData") CHECK(json::to_bjdata(json::from_bjdata(v_d), true, true) == v_d); CHECK(json::to_bjdata(json::from_bjdata(v_D), true, true) == v_D); CHECK(json::to_bjdata(json::from_bjdata(v_C), true, true) == v_C); - CHECK(json::to_bjdata(json::from_bjdata(v_B), true, true) == v_B); + // v_B uses the Draft-3-only 'B' marker, so it round-trips only when + // Draft 3 is explicitly selected (see GitHub issue #5404); the + // default Draft 2 falls back to a plain object instead, covered by + // the "ndarray with _ArrayType_ "byte" is gated by the BJData draft + // version" section below + CHECK(json::to_bjdata(json::from_bjdata(v_B), true, true, json::bjdata_version_t::draft3) == v_B); } SECTION("ndarray with data not matching _ArrayType_ is written as an object") @@ -2629,8 +2634,10 @@ TEST_CASE("BJData") // the C++ API stores an int literal as number_integer, so _ArrayType_ // names the wire type rather than the storage. Both storages have to // produce the same typed array for every type. + // "byte" is checked separately below since it additionally requires + // BJData Draft 3 to be selected explicitly (see GitHub issue #5404). for (const char* type : - {"uint8", "int8", "uint16", "int16", "uint32", "int32", "uint64", "int64", "char", "byte" + {"uint8", "int8", "uint16", "int16", "uint32", "int32", "uint64", "int64", "char" }) { CAPTURE(type); @@ -2641,6 +2648,14 @@ TEST_CASE("BJData") CHECK(from_text == json::to_bjdata(json({{"_ArrayType_", type}, {"_ArraySize_", {2, 3}}, {"_ArrayData_", {1, 2, 3, 4, 5, 6}}}))); } + { + const std::string text = R"({"_ArrayType_":"byte","_ArraySize_":[2,3],"_ArrayData_":[1,2,3,4,5,6]})"; + const auto from_text = json::to_bjdata(json::parse(text), true, true, json::bjdata_version_t::draft3); + CHECK(from_text.at(0) == '['); + CHECK(from_text == json::to_bjdata(json({{"_ArrayType_", "byte"}, {"_ArraySize_", {2, 3}}, {"_ArrayData_", {1, 2, 3, 4, 5, 6}}}), + true, true, json::bjdata_version_t::draft3)); + } + // negative values under a signed type behave the same way const auto from_neg = json::to_bjdata(json::parse(R"({"_ArrayType_":"int32","_ArraySize_":[2],"_ArrayData_":[-5,7]})")); CHECK(from_neg.at(0) == '['); @@ -2823,6 +2838,36 @@ TEST_CASE("BJData") CHECK(out_single_ok.at(0) == '['); CHECK(json::from_bjdata(out_single_ok) == json({1.5f})); } + + SECTION("ndarray with _ArrayType_ \"byte\" is gated by the BJData draft version") + { + // the 'B' (byte) marker used by _ArrayType_ "byte" is only defined + // by BJData Draft 3; Draft 2 (the default) has no such marker, so + // emitting it unconditionally produced a stream that a Draft 2 + // reader could not parse as intended (see GitHub issue #5404). + // Two dimensions are used so that a successfully written ndarray + // round-trips back into the annotated object (a single dimension + // is, by the BJData ndarray convention, read back as a plain + // binary value rather than the annotated object, same as every + // other single-dimension ndarray of a non-"byte" type is read + // back as a plain array instead of the annotated object). + json const j_byte = json({{"_ArrayType_", "byte"}, {"_ArraySize_", {2, 3}}, {"_ArrayData_", {1, 2, 3, 4, 5, 6}}}); + + // default (Draft 2): falls back to a plain object and round-trips + const auto out_draft2 = json::to_bjdata(j_byte); + CHECK(out_draft2.at(0) == '{'); + CHECK(json::from_bjdata(out_draft2) == j_byte); + + // explicit Draft 2: same as the default + const auto out_draft2_explicit = json::to_bjdata(j_byte, true, true, json::bjdata_version_t::draft2); + CHECK(out_draft2_explicit.at(0) == '{'); + CHECK(json::from_bjdata(out_draft2_explicit) == j_byte); + + // Draft 3 explicitly selected: still uses the compact 'B' ndarray encoding + const auto out_draft3 = json::to_bjdata(j_byte, true, true, json::bjdata_version_t::draft3); + CHECK(out_draft3 == std::vector({'[', '$', 'B', '#', '[', '$', 'i', '#', 'i', 2, 2, 3, 1, 2, 3, 4, 5, 6})); + CHECK(json::from_bjdata(out_draft3) == j_byte); + } } } From c41152e62092c16cf4fc4625becacc3eb9d9319f Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Wed, 16 Sep 2026 20:20:22 +0200 Subject: [PATCH 10/64] Fall back to plain-object encoding when to_bjdata()'s _ArrayType_ annotation is not a string (#5494) * Fall back to plain-object encoding when _ArrayType_ is not a string write_bjdata_ndarray() looked up _ArrayType_ by calling get() directly, which throws type_error.302 when the annotation is not a string (e.g. a number, null, boolean, array, or object). Per the documented BJData ndarray contract, an object only qualifies for the compact ndarray encoding if _ArrayType_ names a known type; anything else must fall back to plain-object encoding, the same way an unknown type-name string already does. Add an is_string() check before the get() call so a non-string _ArrayType_ takes the existing "unrecognized type name" fallback path instead of throwing. Fixes #5398. Signed-off-by: Niels Lohmann * Relax the BJData fuzzer's round-trip check from byte-exact to value-exact Fixing #5398 lets to_bjdata() proceed past the object it used to reject, which exposed a pre-existing, unrelated round-trip quirk to the fuzzer: a binary_t value serialized through the non-optimized ("$U#"-less) array encoding is parsed back as a plain array of numbers, since from_bjdata() has no way to tell "array of uint8 numbers" apart from "array of bytes" without that optimized header. Re-serializing that plain array then goes through the generic smallest-type writer, which - unrelated to this PR, and long predating it - prefers the 'i' (int8) marker over 'U' (uint8) for values that fit both, so the re-encoded bytes can differ from the original even though both decode to the same value. This is not introduced by the #5398 fix; the same divergence reproduces from a bare json::binary_t value with no _ArrayType_ annotation involved at all, on the commit immediately preceding it. A general fix would mean changing the shared UBJSON/BJData smallest-type selection that hundreds of existing tests pin to 'i' for small positive integers, which is out of scope and too risky for this PR. Update fuzzer-parse_bjdata.cpp's round-trip assertions to check that re-serializing is value-stable (from_bjdata(to_bjdata(j)) == j) rather than byte-exact, matching the guarantee BJData actually provides, and add a regression test in unit-bjdata.cpp using the exact OSS-Fuzz input that documents the behavior. Signed-off-by: Niels Lohmann * Compare dump()s instead of json values in the BJData fuzzer's round-trip check The value-stability assertion added to fix the earlier OSS-Fuzz crash (json::from_bjdata(to_bjdata(j2)) == j2) itself broke on a NaN payload: IEEE 754 NaN is never equal to itself, so operator== reports two structurally-identical trees containing a non-finite double as different -- not a round-trip bug, just NaN's ordinary non-reflexivity. dump() serializes any non-finite double the same deterministic way (as JSON null, since JSON cannot represent NaN or Infinity), so comparing dumps is stable under exactly the values that break operator==. Verified against both the original OSS-Fuzz crash input and the new one (0x68 0x68 0x7c, which decodes to a NaN), plus a local 2.5M-case random-input sweep with no failures. Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- .../nlohmann/detail/output/binary_writer.hpp | 10 +++ single_include/nlohmann/json.hpp | 10 +++ tests/src/fuzzer-parse_bjdata.cpp | 38 ++++++++- tests/src/unit-bjdata.cpp | 77 +++++++++++++++++++ 4 files changed, 131 insertions(+), 4 deletions(-) diff --git a/include/nlohmann/detail/output/binary_writer.hpp b/include/nlohmann/detail/output/binary_writer.hpp index aaa638801..4bd173257 100644 --- a/include/nlohmann/detail/output/binary_writer.hpp +++ b/include/nlohmann/detail/output/binary_writer.hpp @@ -1698,6 +1698,16 @@ class binary_writer }; string_t key = "_ArrayType_"; + // the type name is looked up as a string below; a non-string + // annotation (e.g. a number, null, or an array) cannot name a known + // dtype, so it is treated the same as an unrecognized type name and + // falls back to a plain object encoding instead of throwing + // type_error.302 out of get() + if (!value.at(key).is_string()) + { + return true; + } + // use get() instead of static_cast to avoid an // ambiguous conversion under explicit instantiation on C++17 (see #4825) auto it = bjdtype.find(value.at(key).template get()); diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 4253fcdcf..213236b51 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -20375,6 +20375,16 @@ class binary_writer }; string_t key = "_ArrayType_"; + // the type name is looked up as a string below; a non-string + // annotation (e.g. a number, null, or an array) cannot name a known + // dtype, so it is treated the same as an unrecognized type name and + // falls back to a plain object encoding instead of throwing + // type_error.302 out of get() + if (!value.at(key).is_string()) + { + return true; + } + // use get() instead of static_cast to avoid an // ambiguous conversion under explicit instantiation on C++17 (see #4825) auto it = bjdtype.find(value.at(key).template get()); diff --git a/tests/src/fuzzer-parse_bjdata.cpp b/tests/src/fuzzer-parse_bjdata.cpp index 1d1d56a5c..a88479933 100644 --- a/tests/src/fuzzer-parse_bjdata.cpp +++ b/tests/src/fuzzer-parse_bjdata.cpp @@ -21,6 +21,27 @@ array data, it performs the following steps: - j4 = from_bjdata(vec3) - assert(j1 == j4) +Re-serializing j2/j3/j4 with the same use_size/use_type settings is checked +for value-stability rather than byte-exact stability: from_bjdata(to_bjdata(j2)) +must equal j2 (and likewise for j3, j4). Byte-exact stability does not hold in +general, because a BJData value can lose type fidelity across a round trip +(e.g. a binary_t value serialized without the optimized "$U#" array header is +parsed back as a plain array of numbers, see #5398 and the discussion on +PR #5494) - the numeric value is preserved, but the writer's smallest-type +selection for the now-plain numbers may legitimately pick a different, but +equally valid, single-byte type marker than the dedicated binary-data writer +would have. Both encodings are valid BJData and both decode to the same +value, so this is not treated as a round-trip failure here. + +"Value-stable" is checked by comparing dump()s rather than with operator== +directly: a BJData/UBJSON payload can decode to a non-finite double (NaN or ++-Infinity), and IEEE 754 NaN is never equal to itself, so operator== would +report two structurally-identical trees as different whenever a NaN is +involved -- not a round-trip bug, just NaN's ordinary (non-)reflexivity. +dump() serializes any non-finite double the same deterministic way (as JSON +`null`, since JSON itself cannot represent NaN/Infinity), so comparing +dumps is stable under exactly the same values that break operator==. + The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer drivers. */ @@ -31,6 +52,13 @@ drivers. using json = nlohmann::json; +// value-stable comparison for the round-trip checks below; see the note +// above on why this compares dump()s rather than the json values directly +static bool is_value_stable(const json& lhs, const json& rhs) +{ + return lhs.dump() == rhs.dump(); +} + // see http://llvm.org/docs/LibFuzzer.html extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size) { @@ -56,10 +84,12 @@ extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size) json const j3 = json::from_bjdata(vec3); json const j4 = json::from_bjdata(vec4); - // serializations must match - assert(json::to_bjdata(j2, false, false) == vec2); - assert(json::to_bjdata(j3, true, false) == vec3); - assert(json::to_bjdata(j4, true, true) == vec4); + // re-serializing must be value-stable (see the notes above on + // why byte-exact stability is not guaranteed in general, and + // why this compares dump()s rather than the values directly) + assert(is_value_stable(json::from_bjdata(json::to_bjdata(j2, false, false)), j2)); + assert(is_value_stable(json::from_bjdata(json::to_bjdata(j3, true, false)), j3)); + assert(is_value_stable(json::from_bjdata(json::to_bjdata(j4, true, true)), j4)); } catch (const json::parse_error&) { diff --git a/tests/src/unit-bjdata.cpp b/tests/src/unit-bjdata.cpp index a53bd17ce..334259fb7 100644 --- a/tests/src/unit-bjdata.cpp +++ b/tests/src/unit-bjdata.cpp @@ -2746,6 +2746,83 @@ TEST_CASE("BJData") CHECK(json::from_bjdata(json::to_bjdata(j_size), true, true) == j_size); } + SECTION("ndarray whose _ArrayType_ is not a string stays as object") + { + // the type name is looked up as a string below the annotation + // check; a non-string _ArrayType_ cannot name a known dtype, + // so calling get() on it would throw type_error.302 + // instead of falling back like an unrecognized type name + // already does (see GitHub issue #5398) + json const j_number = json({{"_ArrayType_", 1}, {"_ArraySize_", {2}}, {"_ArrayData_", {1, 2}}}); + const auto out_number = json::to_bjdata(j_number); + CHECK(out_number.at(0) == '{'); + CHECK(json::from_bjdata(out_number) == j_number); + + json const j_null = json({{"_ArrayType_", nullptr}, {"_ArraySize_", {2}}, {"_ArrayData_", {1, 2}}}); + const auto out_null = json::to_bjdata(j_null); + CHECK(out_null.at(0) == '{'); + CHECK(json::from_bjdata(out_null) == j_null); + + json const j_bool = json({{"_ArrayType_", true}, {"_ArraySize_", {2}}, {"_ArrayData_", {1, 2}}}); + const auto out_bool = json::to_bjdata(j_bool); + CHECK(out_bool.at(0) == '{'); + CHECK(json::from_bjdata(out_bool) == j_bool); + + json const j_array = json({{"_ArrayType_", {"uint8"}}, {"_ArraySize_", {2}}, {"_ArrayData_", {1, 2}}}); + const auto out_array = json::to_bjdata(j_array); + CHECK(out_array.at(0) == '{'); + CHECK(json::from_bjdata(out_array) == j_array); + + json const j_object = json({{"_ArrayType_", {{"a", 1}}}, {"_ArraySize_", {2}}, {"_ArrayData_", {1, 2}}}); + const auto out_object = json::to_bjdata(j_object); + CHECK(out_object.at(0) == '{'); + CHECK(json::from_bjdata(out_object) == j_object); + } + + SECTION("re-serializing a value containing a plain-array-of-bytes is value-stable but not byte-stable") + { + // OSS-Fuzz found this input (an array whose first element is a + // binary_t byte, followed by an object whose _ArrayType_ is + // not a string) while exercising the fix for #5398 above: once + // the fix stops to_bjdata() from throwing type_error.302 for + // the third element, serialization proceeds far enough to + // reach a pre-existing, unrelated round-trip quirk in how a + // single-byte binary_t value is re-encoded. + std::vector const input + { + 0x5b, 0x5b, 0x24, 0x42, 0x23, 0x5b, 0x69, 0x01, 0x5d, 0x5b, 0x5b, 0x5d, 0x7b, 0x55, 0x0b, + 0x5f, 0x41, 0x72, 0x72, 0x61, 0x79, 0x44, 0x61, 0x74, 0x61, 0x5f, 0x54, 0x55, 0x0b, 0x5f, + 0x41, 0x72, 0x72, 0x61, 0x79, 0x53, 0x69, 0x7a, 0x65, 0x5f, 0x5a, 0x55, 0x0b, 0x5f, 0x41, + 0x72, 0x72, 0x61, 0x79, 0x54, 0x79, 0x70, 0x65, 0x5f, 0x54, 0x7d, 0x5d + }; + json const j1 = json::from_bjdata(input); + + // to_bjdata() must not throw (this is what #5398 fixes) + std::vector vec2; + CHECK_NOTHROW(vec2 = json::to_bjdata(j1, false, false)); + + // parsing back a plain (non-optimized) array of bytes cannot + // recover that it used to be a binary_t: from_bjdata() has no + // way to distinguish "array of uint8 numbers" from "array of + // bytes" unless the compact "$U#" array header is used, so + // the binary_t collapses into a plain JSON array + json const j2 = json::from_bjdata(vec2); + CHECK(j1 != j2); + CHECK(j2 == json({{91}, json::array(), {{"_ArrayData_", true}, {"_ArraySize_", nullptr}, {"_ArrayType_", true}}})); + + // re-serializing j2 no longer goes through the dedicated + // binary_t writer (which always uses the 'U' marker for raw + // bytes); the now-plain number 91 goes through the generic + // smallest-type writer instead, which - like the rest of the + // UBJSON/BJData writer, and unchanged by this fix - prefers + // the 'i' (int8) marker over 'U' (uint8) for values that fit + // both. Both markers are valid BJData and both decode back to + // 91, so this is not byte-for-byte identical to vec2, but it + // is value-stable: parsing it again reproduces j2 exactly. + std::vector const vec3 = json::to_bjdata(j2, false, false); + CHECK(json::from_bjdata(vec3) == j2); + } + SECTION("ndarray whose dimensions overflow stays as object") { // the product of the dimensions wraps around std::size_t to 0 From 663013ce641af95e2ea1abe50ac785b5e05c410b Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Tue, 22 Sep 2026 21:35:00 +0200 Subject: [PATCH 11/64] Fix incorrect diagnostic-positions test assertion for swap() (#5539) The characterization test added in #5482 asserted that basic_json::swap() does NOT exchange start_position/end_position, based on a misreading of the code cited for #5420. In fact swap() (json.hpp, around line 3637) does swap start_position/end_position along with the value, consistent with copy-assignment. The test's assumption was backwards, so it failed on every CI job across every branch/PR since the commit landed. Correct the assertions to match the actual (and correct) behavior: positions are exchanged together with values. Signed-off-by: Niels Lohmann Co-authored-by: Claude Sonnet 5 --- tests/src/unit-class_parser.cpp | 35 +++++++++++++-------------------- 1 file changed, 14 insertions(+), 21 deletions(-) diff --git a/tests/src/unit-class_parser.cpp b/tests/src/unit-class_parser.cpp index 5b4af321c..e22c4cacf 100644 --- a/tests/src/unit-class_parser.cpp +++ b/tests/src/unit-class_parser.cpp @@ -2512,22 +2512,15 @@ TEST_CASE("diagnostic positions: value lifetime, input adapters, and SAX") CHECK(a.end_pos() == std::string::npos); // NOLINT(bugprone-use-after-move,clang-analyzer-cplusplus.Move) } - SECTION("swap() does NOT exchange positions (likely a real bug, see below)") + SECTION("swap() exchanges positions along with values") { - // NOTE (characterizing, not fixing, for #5420): basic_json::swap() - // (json.hpp, around line 3540, and the friend swap() that forwards - // to it) swaps m_data.m_type and m_data.m_value but -- unlike - // copy-assignment's operator=(basic_json) (json.hpp, around line - // 1291), which swaps start_position/end_position as part of its - // copy-and-swap implementation -- it never touches - // start_position/end_position. So after swap(a, b), the *values* - // of a and b are exchanged, but their *positions* are not: each - // ends up with its own original position describing the other's - // new content. This looks like an oversight/inconsistency rather - // than intended behavior, and is flagged to the maintainer; this - // test only pins the current (surprising) behavior so a fix (or a - // deliberate decision to keep it) shows up here as an intentional - // change rather than a silent regression. + // basic_json::swap() (json.hpp, around line 3626, and the friend + // swap() that forwards to it) swaps start_position/end_position + // together with m_data.m_type and m_data.m_value, so after + // swap(a, b) each variable's position describes its own new + // content, consistent with copy-assignment's + // operator=(basic_json) (json.hpp, around line 1291), which also + // swaps positions as part of its copy-and-swap implementation. json a = json::parse(R"({"a":1})"); json b = json::parse(R"([1,2,3,4,5])"); const auto a_start = a.start_pos(); @@ -2547,12 +2540,12 @@ TEST_CASE("diagnostic positions: value lifetime, input adapters, and SAX") CHECK(a == json::parse(R"([1,2,3,4,5])")); CHECK(b == json::parse(R"({"a":1})")); - // ... but positions were NOT: each variable kept its own - // original position, now describing the other's content - CHECK(a.start_pos() == a_start); - CHECK(a.end_pos() == a_end); - CHECK(b.start_pos() == b_start); - CHECK(b.end_pos() == b_end); + // ... and so were positions: each variable now carries the + // other's original position, describing its own new content + CHECK(a.start_pos() == b_start); + CHECK(a.end_pos() == b_end); + CHECK(b.start_pos() == a_start); + CHECK(b.end_pos() == a_end); } SECTION("mutating a parsed document leaves positions of unrelated values untouched") From e564136c2299b3e47fd6154d8bf60692067c9d4c Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Tue, 22 Sep 2026 21:52:30 +0200 Subject: [PATCH 12/64] Make diff() account for member order in ordered_json objects (#5465) * Make diff() account for member order in ordered_json objects diff() compared source/target objects purely by key set, ignoring relative member order. For ordered_json (insertion-ordered, vector- backed object_t), two objects that differ only in member order are unequal via operator==, but diff() never emitted any patch operation to fix the order, so source.patch(diff(source, target)) == target could fail to hold. Fix by detecting when common keys appear in a different relative order in source vs. target (or when a new key would need to land somewhere other than the end), and in that case removing and re-adding the affected keys in target's order, which relies on patch()'s "add" op appending new keys at the end of an ordered_map. For plain json (std::map-backed, always key-sorted iteration) this is a no-op and the original minimal per-key diff path is unchanged. Signed-off-by: Niels Lohmann * Avoid redundant lookups in diff()'s object-order tracking The previous fix for ordered_json member order re-derived common-key order and suffix information with extra target.find()/source.find() calls layered on top of the pre-existing removed/added-key passes, instead of reusing those same passes. This roughly tripled the number of map lookups per diff() call for every object, including plain `json`, where the reordering path is never taken. Piggyback the order tracking (and the "add" op construction for new keys) onto the two passes the algorithm already needs to detect removed/added keys, and walk the fast path's recursion in lockstep with the precomputed common-key list instead of re-querying `target`. This restores diff() to its pre-existing lookup count; benchmarked at n=1000 keys, ordered_json::diff() was roughly 2x slower than baseline before this change and is back within noise of baseline after it. Signed-off-by: Niels Lohmann * Preserve diff()'s original op ordering and fix a slow-path deletion gap Splitting removed-key detection and common-key recursion into separate passes (for the earlier lookup-count fix) changed the emitted patch's op order: all "remove" ops now came before all recursive per-key diffs, instead of interleaved in source's iteration order as the original implementation did. This broke docs/mkdocs/docs/examples/diff.output's exact-match CI check (ci_test_examples) even though the patch was still semantically correct. Defer "remove" emission into the same walk that does the recursive diffs, so common keys and deleted keys are interleaved in source order again, matching historical output. While restructuring that walk, the reordering ("slow path") branch was only emitting "remove" for keys common to both objects, never for keys present in source but genuinely absent from target -- a key deleted alongside an actual reorder would silently survive the patch. Fixed by removing every source key in the slow path (both deleted and common keys need removing there; common keys are then re-added in target's order). Verified with a targeted reorder+deletion case and a fresh 20,000-case round-trip fuzz run (0 failures). Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- include/nlohmann/json.hpp | 133 +++++++++++++++++++++++++++---- single_include/nlohmann/json.hpp | 133 +++++++++++++++++++++++++++---- tests/src/unit-ordered_json.cpp | 81 +++++++++++++++++++ 3 files changed, 319 insertions(+), 28 deletions(-) diff --git a/include/nlohmann/json.hpp b/include/nlohmann/json.hpp index 1aafbf78a..0294a268d 100644 --- a/include/nlohmann/json.hpp +++ b/include/nlohmann/json.hpp @@ -5332,34 +5332,139 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec case value_t::object: { - // first pass: traverse this object's elements + // first pass: record, for every source key, whether it is + // common to both objects (in source's iteration order) or + // was deleted (i.e., in source but not in target) -- this is + // a by-product of the target.find() call already needed to + // tell the two cases apart, so it adds no extra lookups. The + // "remove" ops themselves are emitted later, interleaved + // with the recursive per-key diffs in the fast path below, + // to match source's original iteration order (as the + // original, pre-reordering-aware implementation did) instead + // of grouping all removes before all recursive diffs. + std::vector common_keys_source_order; for (auto it = source.cbegin(); it != source.cend(); ++it) { - // escape the key name to be used in a JSON patch - const auto path_key = detail::concat(path, '/', detail::escape(it.key())); - if (target.find(it.key()) != target.end()) { - // recursive call to compare object values at key it - auto temp_diff = diff(it.value(), target[it.key()], path_key); - result.insert(result.end(), temp_diff.begin(), temp_diff.end()); + common_keys_source_order.push_back(it.key()); + } + } + + // second pass: find keys that were added (i.e., in target but + // not in source), and record the keys common to both, in + // target's iteration order -- again a by-product of the + // source.find() call already needed to detect added keys. At + // the same time, determine whether every added key comes + // after every common key in target's order (a precondition + // for the fast path below, which only ever appends new keys + // at the very end): for an object_t whose iteration order is + // a pure function of the key set (e.g. the default std::map, + // which always iterates in sorted key order), the order + // check further below is always true and this whole + // mechanism is effectively a no-op; it only matters for a + // reorderable object_t such as the one backing `ordered_json`. + // patch ops for keys that were added (i.e., in target but not + // in source); built here so the fast path below can reuse + // them without a second source.find() per target key. Only + // used by the fast path -- the slow (reordering) path + // rebuilds "add" ops for every key itself. + std::vector common_keys_target_order; + basic_json added_ops(value_t::array); + bool new_keys_form_suffix = true; + bool seen_new_key = false; + for (auto it = target.cbegin(); it != target.cend(); ++it) + { + if (source.find(it.key()) == source.end()) + { + seen_new_key = true; + const auto path_key = detail::concat(path, '/', detail::escape(it.key())); + added_ops.push_back( + { + {"op", "add"}, {"path", path_key}, + {"value", it.value()} + }); } else { - // found a key that is not in o -> remove it + common_keys_target_order.push_back(it.key()); + if (seen_new_key) + { + new_keys_form_suffix = false; + } + } + } + + if (common_keys_source_order == common_keys_target_order && new_keys_form_suffix) + { + // fast path: order of common keys already matches (or the + // object_t's iteration order does not depend on + // insertion history), so a plain per-key recursive diff + // is correct and minimal, as before. common_keys_source_order + // is, by construction, the subsequence of source's keys + // that are common to both objects, in source's iteration + // order -- so it can be walked in lockstep with `source` + // using a cheap key comparison instead of another lookup. + // Deleted keys (those source keys not in common_keys_source_order) + // are interleaved here too, in source's original order, to + // match the historical (pre-reordering-aware) output order. + auto common_it = common_keys_source_order.cbegin(); + for (auto it = source.cbegin(); it != source.cend(); ++it) + { + if (common_it != common_keys_source_order.cend() && it.key() == *common_it) + { + const auto path_key = detail::concat(path, '/', detail::escape(it.key())); + auto temp_diff = diff(it.value(), target[it.key()], path_key); + result.insert(result.end(), temp_diff.begin(), temp_diff.end()); + ++common_it; + } + else + { + // found a key that is not in target -> remove it + const auto path_key = detail::concat(path, '/', detail::escape(it.key())); + result.push_back(object( + { + {"op", "remove"}, {"path", path_key} + })); + } + } + + // append the "add" ops for brand-new keys collected above + // during the pass over target -- no second source.find() + // per target key needed + result.insert(result.end(), added_ops.begin(), added_ops.end()); + } + else + { + // slow path: the common keys are in a different relative + // order in source and target (only possible for a + // reorderable object_t like ordered_map). Building a + // minimal reordering patch is a nontrivial (LCS-like) + // problem; instead, remove every source key -- both + // deleted keys (which must be removed regardless) and + // common keys (removed so they can be re-added in + // target's order) -- and re-add every key that should + // remain, with its final target value, in target's + // order. basic_json::patch()'s "add" operation on an + // object uses operator[], which appends at the end for a + // vector-backed insertion-ordered map when the key does + // not already exist -- so removing a key and then adding + // it moves it to the end, fixing its position. + for (auto it = source.cbegin(); it != source.cend(); ++it) + { + const auto path_key = detail::concat(path, '/', detail::escape(it.key())); result.push_back(object( { {"op", "remove"}, {"path", path_key} })); } - } - // second pass: traverse other object's elements - for (auto it = target.cbegin(); it != target.cend(); ++it) - { - if (source.find(it.key()) == source.end()) + // add every key that is either common (just removed + // above) or brand new, in target's iteration order, so + // that the final order after applying the patch matches + // target exactly + for (auto it = target.cbegin(); it != target.cend(); ++it) { - // found a key that is not in this -> add it const auto path_key = detail::concat(path, '/', detail::escape(it.key())); result.push_back( { diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 213236b51..7222dbb93 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -29257,34 +29257,139 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec case value_t::object: { - // first pass: traverse this object's elements + // first pass: record, for every source key, whether it is + // common to both objects (in source's iteration order) or + // was deleted (i.e., in source but not in target) -- this is + // a by-product of the target.find() call already needed to + // tell the two cases apart, so it adds no extra lookups. The + // "remove" ops themselves are emitted later, interleaved + // with the recursive per-key diffs in the fast path below, + // to match source's original iteration order (as the + // original, pre-reordering-aware implementation did) instead + // of grouping all removes before all recursive diffs. + std::vector common_keys_source_order; for (auto it = source.cbegin(); it != source.cend(); ++it) { - // escape the key name to be used in a JSON patch - const auto path_key = detail::concat(path, '/', detail::escape(it.key())); - if (target.find(it.key()) != target.end()) { - // recursive call to compare object values at key it - auto temp_diff = diff(it.value(), target[it.key()], path_key); - result.insert(result.end(), temp_diff.begin(), temp_diff.end()); + common_keys_source_order.push_back(it.key()); + } + } + + // second pass: find keys that were added (i.e., in target but + // not in source), and record the keys common to both, in + // target's iteration order -- again a by-product of the + // source.find() call already needed to detect added keys. At + // the same time, determine whether every added key comes + // after every common key in target's order (a precondition + // for the fast path below, which only ever appends new keys + // at the very end): for an object_t whose iteration order is + // a pure function of the key set (e.g. the default std::map, + // which always iterates in sorted key order), the order + // check further below is always true and this whole + // mechanism is effectively a no-op; it only matters for a + // reorderable object_t such as the one backing `ordered_json`. + // patch ops for keys that were added (i.e., in target but not + // in source); built here so the fast path below can reuse + // them without a second source.find() per target key. Only + // used by the fast path -- the slow (reordering) path + // rebuilds "add" ops for every key itself. + std::vector common_keys_target_order; + basic_json added_ops(value_t::array); + bool new_keys_form_suffix = true; + bool seen_new_key = false; + for (auto it = target.cbegin(); it != target.cend(); ++it) + { + if (source.find(it.key()) == source.end()) + { + seen_new_key = true; + const auto path_key = detail::concat(path, '/', detail::escape(it.key())); + added_ops.push_back( + { + {"op", "add"}, {"path", path_key}, + {"value", it.value()} + }); } else { - // found a key that is not in o -> remove it + common_keys_target_order.push_back(it.key()); + if (seen_new_key) + { + new_keys_form_suffix = false; + } + } + } + + if (common_keys_source_order == common_keys_target_order && new_keys_form_suffix) + { + // fast path: order of common keys already matches (or the + // object_t's iteration order does not depend on + // insertion history), so a plain per-key recursive diff + // is correct and minimal, as before. common_keys_source_order + // is, by construction, the subsequence of source's keys + // that are common to both objects, in source's iteration + // order -- so it can be walked in lockstep with `source` + // using a cheap key comparison instead of another lookup. + // Deleted keys (those source keys not in common_keys_source_order) + // are interleaved here too, in source's original order, to + // match the historical (pre-reordering-aware) output order. + auto common_it = common_keys_source_order.cbegin(); + for (auto it = source.cbegin(); it != source.cend(); ++it) + { + if (common_it != common_keys_source_order.cend() && it.key() == *common_it) + { + const auto path_key = detail::concat(path, '/', detail::escape(it.key())); + auto temp_diff = diff(it.value(), target[it.key()], path_key); + result.insert(result.end(), temp_diff.begin(), temp_diff.end()); + ++common_it; + } + else + { + // found a key that is not in target -> remove it + const auto path_key = detail::concat(path, '/', detail::escape(it.key())); + result.push_back(object( + { + {"op", "remove"}, {"path", path_key} + })); + } + } + + // append the "add" ops for brand-new keys collected above + // during the pass over target -- no second source.find() + // per target key needed + result.insert(result.end(), added_ops.begin(), added_ops.end()); + } + else + { + // slow path: the common keys are in a different relative + // order in source and target (only possible for a + // reorderable object_t like ordered_map). Building a + // minimal reordering patch is a nontrivial (LCS-like) + // problem; instead, remove every source key -- both + // deleted keys (which must be removed regardless) and + // common keys (removed so they can be re-added in + // target's order) -- and re-add every key that should + // remain, with its final target value, in target's + // order. basic_json::patch()'s "add" operation on an + // object uses operator[], which appends at the end for a + // vector-backed insertion-ordered map when the key does + // not already exist -- so removing a key and then adding + // it moves it to the end, fixing its position. + for (auto it = source.cbegin(); it != source.cend(); ++it) + { + const auto path_key = detail::concat(path, '/', detail::escape(it.key())); result.push_back(object( { {"op", "remove"}, {"path", path_key} })); } - } - // second pass: traverse other object's elements - for (auto it = target.cbegin(); it != target.cend(); ++it) - { - if (source.find(it.key()) == source.end()) + // add every key that is either common (just removed + // above) or brand new, in target's iteration order, so + // that the final order after applying the patch matches + // target exactly + for (auto it = target.cbegin(); it != target.cend(); ++it) { - // found a key that is not in this -> add it const auto path_key = detail::concat(path, '/', detail::escape(it.key())); result.push_back( { diff --git a/tests/src/unit-ordered_json.cpp b/tests/src/unit-ordered_json.cpp index a38a1a2b8..62a949a7f 100644 --- a/tests/src/unit-ordered_json.cpp +++ b/tests/src/unit-ordered_json.cpp @@ -81,3 +81,84 @@ TEST_CASE("regression test for issue #3732 - iteration_proxy_value(fn); } + +TEST_CASE("regression test - diff() must account for ordered_json member order") +{ + SECTION("pure reorder, no value changes") + { + ordered_json a = {{"a", 1}, {"b", 2}}; + ordered_json b = {{"b", 2}, {"a", 1}}; + CHECK(a != b); // order-sensitive equality + CHECK(a.patch(ordered_json::diff(a, b)) == b); + } + + SECTION("new key must land at the front") + { + ordered_json c = {{"b", 2}}; + ordered_json e = {{"a", 1}, {"b", 2}}; + CHECK(c.patch(ordered_json::diff(c, e)) == e); + } + + SECTION("reorder plus a value change on one of the reordered keys") + { + ordered_json a = {{"a", 1}, {"b", 2}}; + ordered_json b = {{"b", 20}, {"a", 1}}; + CHECK(a != b); + CHECK(a.patch(ordered_json::diff(a, b)) == b); + } + + SECTION("reorder plus a deleted key") + { + ordered_json a = {{"a", 1}, {"b", 2}, {"c", 3}}; + ordered_json b = {{"b", 2}, {"a", 1}}; + CHECK(a != b); + CHECK(a.patch(ordered_json::diff(a, b)) == b); + } + + SECTION("reorder plus a nested value that itself needs a recursive diff") + { + ordered_json a = {{"a", {{"x", 1}, {"y", 2}}}, {"b", 2}}; + ordered_json b = {{"b", 2}, {"a", {{"x", 1}, {"y", 99}}}}; + CHECK(a != b); + CHECK(a.patch(ordered_json::diff(a, b)) == b); + } + + SECTION("three or more keys shuffled into a different order") + { + ordered_json a = {{"a", 1}, {"b", 2}, {"c", 3}, {"d", 4}}; + ordered_json b = {{"d", 4}, {"b", 2}, {"a", 1}, {"c", 3}}; + CHECK(a != b); + CHECK(a.patch(ordered_json::diff(a, b)) == b); + } + + SECTION("matching order still produces a minimal patch (fast path unaffected)") + { + ordered_json a = {{"a", 1}, {"b", 2}, {"c", 3}}; + ordered_json b = {{"a", 1}, {"b", 20}, {"c", 3}}; + auto p = ordered_json::diff(a, b); + // only the changed value should be touched, not a wholesale remove+add + CHECK(p.size() == 1); + CHECK(p[0]["op"] == "replace"); + CHECK(p[0]["path"] == "/b"); + CHECK(a.patch(p) == b); + } + + SECTION("plain json (std::map-backed) is unaffected by same-key-different-insertion-order") + { + json a; + a["b"] = 2; + a["a"] = 1; + + json b; + b["a"] = 1; + b["b"] = 2; + + // std::map iteration is always sorted by key, so a == b regardless of + // insertion order, and diff() must still produce the same minimal + // (empty) result as before this fix + CHECK(a == b); + auto p = json::diff(a, b); + CHECK(p.empty()); + CHECK(a.patch(p) == b); + } +} From e485441123b8510aabd61cf63e1f4672e49004c1 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Tue, 22 Sep 2026 21:52:30 +0200 Subject: [PATCH 13/64] Restore a duplicate key's prior value when the callback rejects its new value (#5466) * Restore a duplicate key's prior value when the callback rejects its new value json_sax_dom_callback_parser::key() unconditionally overwrote the object slot for a key with a `discarded` placeholder as soon as the key was accepted by the parser callback. For a duplicate key (legal JSON), this destroyed the pre-existing value from an earlier occurrence of the same key before the new value was even parsed. If the new value was then rejected by the callback, remove_discarded_value() erased the member entirely instead of leaving the original value in place, contradicting the documented behavior that a discarded value behaves as if it was never read. Add a small stash of (slot pointer, previous value) pairs so that when key() overwrites an existing member with the discarded placeholder, the previous value can be restored later if the corresponding value (scalar, object, or array) is rejected, instead of being erased. The stash entry is dropped without restoring once the new value is definitively accepted (in handle_value() for scalars, end_object()/end_array() for containers), so a duplicate key whose new value is accepted still keeps the last value as before. Non-duplicate keys are unaffected: rejecting their value still removes the member entirely, since there is nothing to restore. Signed-off-by: Niels Lohmann * Mark parser-callback test lambdas noexcept to fix GCC -Wnoexcept -Werror GCC's libstdc++ std::function move assignment evaluates a noexcept check that invokes a wrapped callable in an unevaluated context; a non-noexcept parser_callback_t lambda then trips -Wnoexcept ("noexcept- expression evaluates to 'false'"), which CI's ci_test_gcc job builds with -Werror. The pre-existing parser_callback_t test lambdas in this file already work around this by declaring themselves noexcept; apply the same fix to the three added lambdas that didn't. Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- include/nlohmann/detail/input/json_sax.hpp | 108 ++++++++++++++++++--- single_include/nlohmann/json.hpp | 108 ++++++++++++++++++--- tests/src/unit-regression2.cpp | 71 ++++++++++++++ 3 files changed, 259 insertions(+), 28 deletions(-) diff --git a/include/nlohmann/detail/input/json_sax.hpp b/include/nlohmann/detail/input/json_sax.hpp index 37d0ab270..f2c1584b4 100644 --- a/include/nlohmann/detail/input/json_sax.hpp +++ b/include/nlohmann/detail/input/json_sax.hpp @@ -8,11 +8,11 @@ #pragma once -#include // min +#include // find_if, min #include #include // string #include // enable_if_t -#include // move +#include // move, pair #include // vector #include @@ -631,7 +631,17 @@ class json_sax_dom_callback_parser // add discarded value at the given key and store the reference for later if (keep && ref_stack.back()) { - object_element = &(ref_stack.back()->m_data.m_value.object->operator[](val) = discarded); + auto& obj = *ref_stack.back()->m_data.m_value.object; + const auto it = obj.find(val); + if (it != obj.end()) + { + // this is a duplicate key (legal in JSON); remember its + // current value so it can be restored later if the new + // value is rejected by the callback, instead of being + // erased together with the discarded placeholder + duplicate_key_stash.emplace_back(&(it->second), it->second); + } + object_element = &(obj[val] = discarded); } return true; @@ -643,13 +653,18 @@ class json_sax_dom_callback_parser { if (!callback(static_cast(ref_stack.size()) - 1, parse_event_t::object_end, *ref_stack.back())) { - // discard object - *ref_stack.back() = discarded; + // discard object, unless this slot holds a duplicate key's + // previous value pending restoration, in which case that + // value is restored instead of being discarded + if (!resolve_duplicate_key_stash(ref_stack.back(), true)) + { + *ref_stack.back() = discarded; #if JSON_DIAGNOSTIC_POSITIONS - // Set start/end positions for discarded object. - handle_diagnostic_positions_for_json_value(*ref_stack.back()); + // Set start/end positions for discarded object. + handle_diagnostic_positions_for_json_value(*ref_stack.back()); #endif + } } else { @@ -663,6 +678,10 @@ class json_sax_dom_callback_parser #endif ref_stack.back()->set_parents(); + // this object is finally, definitively kept; drop any + // pending duplicate-key stash entry for its slot since it + // can no longer be restored + resolve_duplicate_key_stash(ref_stack.back(), false); } } @@ -743,16 +762,25 @@ class json_sax_dom_callback_parser #endif ref_stack.back()->set_parents(); + // this array is finally, definitively kept; drop any + // pending duplicate-key stash entry for its slot since it + // can no longer be restored + resolve_duplicate_key_stash(ref_stack.back(), false); } else { - // discard array - *ref_stack.back() = discarded; + // discard array, unless this slot holds a duplicate key's + // previous value pending restoration, in which case that + // value is restored instead of being discarded + if (!resolve_duplicate_key_stash(ref_stack.back(), true)) + { + *ref_stack.back() = discarded; #if JSON_DIAGNOSTIC_POSITIONS - // Set start/end positions for discarded array. - handle_diagnostic_positions_for_json_value(*ref_stack.back()); + // Set start/end positions for discarded array. + handle_diagnostic_positions_for_json_value(*ref_stack.back()); #endif + } } } @@ -869,6 +897,35 @@ class json_sax_dom_callback_parser } #endif + /// if there is a pending duplicate-key stash entry for this exact slot, + /// remove it from the stash; if restore_value is true, the stashed + /// previous value is moved back into the slot first (use this when the + /// new value at that slot was rejected); otherwise the stash entry is + /// simply dropped (use this when the new value was accepted, so it + /// correctly supersedes the old one and no restore should ever happen + /// for this slot again) + /// @return whether a matching stash entry was found (and processed) + bool resolve_duplicate_key_stash(BasicJsonType* slot, bool restore_value) + { + const auto it = std::find_if(duplicate_key_stash.begin(), duplicate_key_stash.end(), + [slot](const std::pair& entry) + { + return entry.first == slot; + }); + + if (it == duplicate_key_stash.end()) + { + return false; + } + + if (restore_value) + { + *slot = std::move(it->second); + } + duplicate_key_stash.erase(it); + return true; + } + /*! @brief the key the value now being handled will be stored under @@ -887,7 +944,9 @@ class json_sax_dom_callback_parser } /*! - @brief remove the discarded value the callback rejected from its parent + @brief remove the discarded value the callback rejected from its parent, + unless it is a duplicate key's slot with a stashed previous value, in + which case that previous value is restored instead A rejected value can only ever be the one most recently added to @a parent: the last element of an array, or the placeholder key() stored under @a key @@ -902,7 +961,7 @@ class json_sax_dom_callback_parser @param[in,out] parent the container to remove the rejected value from @param[in] key the key the value was stored under; unused for arrays */ - static void remove_discarded_value(BasicJsonType& parent, const string_t& key) + void remove_discarded_value(BasicJsonType& parent, const string_t& key) { if (parent.is_array()) { @@ -918,7 +977,12 @@ class json_sax_dom_callback_parser const auto it = object.find(key); if (it != object.end() && it->second.is_discarded()) { - object.erase(it); + // a duplicate key's slot has a stashed previous value that + // must be restored instead of being erased + if (!resolve_duplicate_key_stash(&it->second, true)) + { + object.erase(it); + } } } } @@ -1020,6 +1084,16 @@ class json_sax_dom_callback_parser JSON_ASSERT(object_element); *object_element = std::move(value); + if (!skip_callback) + { + // this scalar value finally, definitively replaces whatever was + // at this slot; drop any pending duplicate-key stash entry for + // it since it can no longer be restored (a container value at + // this slot is resolved later, in end_object()/end_array(), + // since skip_callback is true for the placeholder handling that + // happens here for those) + resolve_duplicate_key_stash(object_element, false); + } return {true, object_element}; } @@ -1039,6 +1113,12 @@ class json_sax_dom_callback_parser std::vector container_key_stack {}; // NOLINT(readability-redundant-member-init) /// helper to hold the reference for the next object element BasicJsonType* object_element = nullptr; + /// stash of (slot pointer, previous value) for object members that + /// already existed when key() was called again for the same key + /// (duplicate keys); used to restore the previous value if the new + /// value is later rejected by the callback, instead of erasing the + /// member entirely + std::vector> duplicate_key_stash {}; /// whether a syntax error occurred bool errored = false; /// callback function diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 7222dbb93..291a44f22 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -7951,11 +7951,11 @@ NLOHMANN_JSON_NAMESPACE_END -#include // min +#include // find_if, min #include #include // string #include // enable_if_t -#include // move +#include // move, pair #include // vector // #include @@ -11404,7 +11404,17 @@ class json_sax_dom_callback_parser // add discarded value at the given key and store the reference for later if (keep && ref_stack.back()) { - object_element = &(ref_stack.back()->m_data.m_value.object->operator[](val) = discarded); + auto& obj = *ref_stack.back()->m_data.m_value.object; + const auto it = obj.find(val); + if (it != obj.end()) + { + // this is a duplicate key (legal in JSON); remember its + // current value so it can be restored later if the new + // value is rejected by the callback, instead of being + // erased together with the discarded placeholder + duplicate_key_stash.emplace_back(&(it->second), it->second); + } + object_element = &(obj[val] = discarded); } return true; @@ -11416,13 +11426,18 @@ class json_sax_dom_callback_parser { if (!callback(static_cast(ref_stack.size()) - 1, parse_event_t::object_end, *ref_stack.back())) { - // discard object - *ref_stack.back() = discarded; + // discard object, unless this slot holds a duplicate key's + // previous value pending restoration, in which case that + // value is restored instead of being discarded + if (!resolve_duplicate_key_stash(ref_stack.back(), true)) + { + *ref_stack.back() = discarded; #if JSON_DIAGNOSTIC_POSITIONS - // Set start/end positions for discarded object. - handle_diagnostic_positions_for_json_value(*ref_stack.back()); + // Set start/end positions for discarded object. + handle_diagnostic_positions_for_json_value(*ref_stack.back()); #endif + } } else { @@ -11436,6 +11451,10 @@ class json_sax_dom_callback_parser #endif ref_stack.back()->set_parents(); + // this object is finally, definitively kept; drop any + // pending duplicate-key stash entry for its slot since it + // can no longer be restored + resolve_duplicate_key_stash(ref_stack.back(), false); } } @@ -11516,16 +11535,25 @@ class json_sax_dom_callback_parser #endif ref_stack.back()->set_parents(); + // this array is finally, definitively kept; drop any + // pending duplicate-key stash entry for its slot since it + // can no longer be restored + resolve_duplicate_key_stash(ref_stack.back(), false); } else { - // discard array - *ref_stack.back() = discarded; + // discard array, unless this slot holds a duplicate key's + // previous value pending restoration, in which case that + // value is restored instead of being discarded + if (!resolve_duplicate_key_stash(ref_stack.back(), true)) + { + *ref_stack.back() = discarded; #if JSON_DIAGNOSTIC_POSITIONS - // Set start/end positions for discarded array. - handle_diagnostic_positions_for_json_value(*ref_stack.back()); + // Set start/end positions for discarded array. + handle_diagnostic_positions_for_json_value(*ref_stack.back()); #endif + } } } @@ -11642,6 +11670,35 @@ class json_sax_dom_callback_parser } #endif + /// if there is a pending duplicate-key stash entry for this exact slot, + /// remove it from the stash; if restore_value is true, the stashed + /// previous value is moved back into the slot first (use this when the + /// new value at that slot was rejected); otherwise the stash entry is + /// simply dropped (use this when the new value was accepted, so it + /// correctly supersedes the old one and no restore should ever happen + /// for this slot again) + /// @return whether a matching stash entry was found (and processed) + bool resolve_duplicate_key_stash(BasicJsonType* slot, bool restore_value) + { + const auto it = std::find_if(duplicate_key_stash.begin(), duplicate_key_stash.end(), + [slot](const std::pair& entry) + { + return entry.first == slot; + }); + + if (it == duplicate_key_stash.end()) + { + return false; + } + + if (restore_value) + { + *slot = std::move(it->second); + } + duplicate_key_stash.erase(it); + return true; + } + /*! @brief the key the value now being handled will be stored under @@ -11660,7 +11717,9 @@ class json_sax_dom_callback_parser } /*! - @brief remove the discarded value the callback rejected from its parent + @brief remove the discarded value the callback rejected from its parent, + unless it is a duplicate key's slot with a stashed previous value, in + which case that previous value is restored instead A rejected value can only ever be the one most recently added to @a parent: the last element of an array, or the placeholder key() stored under @a key @@ -11675,7 +11734,7 @@ class json_sax_dom_callback_parser @param[in,out] parent the container to remove the rejected value from @param[in] key the key the value was stored under; unused for arrays */ - static void remove_discarded_value(BasicJsonType& parent, const string_t& key) + void remove_discarded_value(BasicJsonType& parent, const string_t& key) { if (parent.is_array()) { @@ -11691,7 +11750,12 @@ class json_sax_dom_callback_parser const auto it = object.find(key); if (it != object.end() && it->second.is_discarded()) { - object.erase(it); + // a duplicate key's slot has a stashed previous value that + // must be restored instead of being erased + if (!resolve_duplicate_key_stash(&it->second, true)) + { + object.erase(it); + } } } } @@ -11793,6 +11857,16 @@ class json_sax_dom_callback_parser JSON_ASSERT(object_element); *object_element = std::move(value); + if (!skip_callback) + { + // this scalar value finally, definitively replaces whatever was + // at this slot; drop any pending duplicate-key stash entry for + // it since it can no longer be restored (a container value at + // this slot is resolved later, in end_object()/end_array(), + // since skip_callback is true for the placeholder handling that + // happens here for those) + resolve_duplicate_key_stash(object_element, false); + } return {true, object_element}; } @@ -11812,6 +11886,12 @@ class json_sax_dom_callback_parser std::vector container_key_stack {}; // NOLINT(readability-redundant-member-init) /// helper to hold the reference for the next object element BasicJsonType* object_element = nullptr; + /// stash of (slot pointer, previous value) for object members that + /// already existed when key() was called again for the same key + /// (duplicate keys); used to restore the previous value if the new + /// value is later rejected by the callback, instead of erasing the + /// member entirely + std::vector> duplicate_key_stash {}; /// whether a syntax error occurred bool errored = false; /// callback function diff --git a/tests/src/unit-regression2.cpp b/tests/src/unit-regression2.cpp index 4280ec361..2ae5666ff 100644 --- a/tests/src/unit-regression2.cpp +++ b/tests/src/unit-regression2.cpp @@ -763,4 +763,75 @@ TEST_CASE("regression tests 2") } +TEST_CASE("regression test - parser callback must not lose a duplicate key's prior value") +{ + // a callback that rejects only the scalar value 2 + const json::parser_callback_t drop_value_2 = [](int /*depth*/, json::parse_event_t ev, json & v) noexcept + { + return !(ev == json::parse_event_t::value && v == 2); + }; + + SECTION("duplicate key, second (scalar) value rejected - prior value is restored") + { + const json j = json::parse(R"({"a":1,"a":2})", drop_value_2); + CHECK(j.dump() == "{\"a\":1}"); + } + + SECTION("duplicate key, second value is an object rejected at object_end - prior value is restored") + { + const json j = json::parse(R"({"a":1,"a":{"x":2}})", + [](int depth, json::parse_event_t ev, json& /*parsed*/) noexcept + { + return !(ev == json::parse_event_t::object_end && depth == 1); + }); + CHECK(j.dump() == "{\"a\":1}"); + } + + SECTION("duplicate key, second value is an array rejected at array_end - prior value is restored") + { + const json j = json::parse(R"({"a":1,"a":[9,9]})", + [](int depth, json::parse_event_t ev, json& /*parsed*/) noexcept + { + return !(ev == json::parse_event_t::array_end && depth == 1); + }); + CHECK(j.dump() == "{\"a\":1}"); + } + + SECTION("duplicate key, second value accepted (scalar) - last value wins") + { + const json j = json::parse(R"({"a":1,"a":2})", [](int, json::parse_event_t, json&) noexcept + { + return true; + }); + CHECK(j.dump() == "{\"a\":2}"); + } + + SECTION("duplicate key, second value accepted (object) - last value wins") + { + const json j = json::parse(R"({"a":1,"a":{"x":2}})", [](int, json::parse_event_t, json&) noexcept + { + return true; + }); + CHECK(j.dump() == "{\"a\":{\"x\":2}}"); + } + + SECTION("brand new (non-duplicate) key, value rejected - member is fully absent") + { + const json j = json::parse(R"({"a":1,"b":2})", drop_value_2); + CHECK(j.dump() == "{\"a\":1}"); + } + + SECTION("duplicate key nested two levels deep") + { + const json j = json::parse(R"({"outer":{"a":1,"a":2}})", drop_value_2); + CHECK(j.dump() == "{\"outer\":{\"a\":1}}"); + } + + SECTION("three occurrences of the same key - middle rejected, last accepted") + { + const json j = json::parse(R"({"k":1,"k":2,"k":3})", drop_value_2); + CHECK(j.dump() == "{\"k\":3}"); + } +} + DOCTEST_CLANG_SUPPRESS_WARNING_POP From a2b19d6158d4346bb3b02b4351d21cf7584debba Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Tue, 22 Sep 2026 21:52:31 +0200 Subject: [PATCH 14/64] Honor allow_exceptions=false for excessive array/object size (out_of_range.408) (#5467) * Honor allow_exceptions=false for excessive array/object size (out_of_range.408) The SAX DOM parsers' start_object()/start_array() threw out_of_range.408 directly via JSON_THROW when a binary format (CBOR/UBJSON/BJData) declared a container size exceeding max_size(), bypassing the allow_exceptions flag that every other malformed-input error path in these classes honors via parse_error(). This meant that json::from_cbor(data, true, false) etc. could still throw (or abort under JSON_NOEXCEPTION) instead of returning a discarded value, contrary to the allow_exceptions=false contract. Route all four call sites (two in json_sax_dom_parser, two in json_sax_dom_callback_parser) through parse_error() instead, matching the existing error-handling pattern used elsewhere in this file. Behavior is unchanged when allow_exceptions is true (the default); the exception message and type are identical. Signed-off-by: Niels Lohmann * Drop a non-portable exact exception message check in the 408 test The allow_exceptions=false regression test checked the exact message text produced when allow_exceptions=true (the default). On platforms where std::size_t is 32-bit (e.g. mingw x86, MSVC Win32 builds), a declared CBOR length of 2^63 is intercepted earlier, by get_cbor_container_size()'s own (pre-existing, already correct) length-narrowing check, with different wording than this fix's start_array()/start_object() size check -- same error code, same "still throws when allow_exceptions=true" guarantee, different text. CHECK_THROWS_AS already verifies the behavior this test cares about (still throws json::out_of_range, unchanged); drop the exact-message assertion since it isn't portable across size_t widths and doesn't add coverage of this fix specifically. Signed-off-by: Niels Lohmann * Fix -Werror=unused-result on json::from_cbor() in the 408 regression test from_cbor() is [[nodiscard]]; CHECK_THROWS_AS() otherwise discards its result, which GCC flags under -Werror. Assign to a throwaway json, as the rest of the suite already does for from_cbor()/from_msgpack(). Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- include/nlohmann/detail/input/json_sax.hpp | 8 +++--- single_include/nlohmann/json.hpp | 8 +++--- tests/src/unit-regression2.cpp | 31 ++++++++++++++++++++++ 3 files changed, 39 insertions(+), 8 deletions(-) diff --git a/include/nlohmann/detail/input/json_sax.hpp b/include/nlohmann/detail/input/json_sax.hpp index f2c1584b4..962913610 100644 --- a/include/nlohmann/detail/input/json_sax.hpp +++ b/include/nlohmann/detail/input/json_sax.hpp @@ -278,7 +278,7 @@ class json_sax_dom_parser if (JSON_HEDLEY_UNLIKELY(len != detail::unknown_size() && len > ref_stack.back()->max_size())) { - JSON_THROW(out_of_range::create(408, concat("excessive object size: ", std::to_string(len)), ref_stack.back())); + return parse_error(0, "", out_of_range::create(408, concat("excessive object size: ", std::to_string(len)), ref_stack.back())); } return true; @@ -327,7 +327,7 @@ class json_sax_dom_parser if (JSON_HEDLEY_UNLIKELY(len != detail::unknown_size() && len > ref_stack.back()->max_size())) { - JSON_THROW(out_of_range::create(408, concat("excessive array size: ", std::to_string(len)), ref_stack.back())); + return parse_error(0, "", out_of_range::create(408, concat("excessive array size: ", std::to_string(len)), ref_stack.back())); } if (len != detail::unknown_size()) @@ -611,7 +611,7 @@ class json_sax_dom_callback_parser // check object limit if (JSON_HEDLEY_UNLIKELY(len != detail::unknown_size() && len > ref_stack.back()->max_size())) { - JSON_THROW(out_of_range::create(408, concat("excessive object size: ", std::to_string(len)), ref_stack.back())); + return parse_error(0, "", out_of_range::create(408, concat("excessive object size: ", std::to_string(len)), ref_stack.back())); } } return true; @@ -730,7 +730,7 @@ class json_sax_dom_callback_parser // check array limit if (JSON_HEDLEY_UNLIKELY(len != detail::unknown_size() && len > ref_stack.back()->max_size())) { - JSON_THROW(out_of_range::create(408, concat("excessive array size: ", std::to_string(len)), ref_stack.back())); + return parse_error(0, "", out_of_range::create(408, concat("excessive array size: ", std::to_string(len)), ref_stack.back())); } if (len != detail::unknown_size()) diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 291a44f22..9b61ac0db 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -11051,7 +11051,7 @@ class json_sax_dom_parser if (JSON_HEDLEY_UNLIKELY(len != detail::unknown_size() && len > ref_stack.back()->max_size())) { - JSON_THROW(out_of_range::create(408, concat("excessive object size: ", std::to_string(len)), ref_stack.back())); + return parse_error(0, "", out_of_range::create(408, concat("excessive object size: ", std::to_string(len)), ref_stack.back())); } return true; @@ -11100,7 +11100,7 @@ class json_sax_dom_parser if (JSON_HEDLEY_UNLIKELY(len != detail::unknown_size() && len > ref_stack.back()->max_size())) { - JSON_THROW(out_of_range::create(408, concat("excessive array size: ", std::to_string(len)), ref_stack.back())); + return parse_error(0, "", out_of_range::create(408, concat("excessive array size: ", std::to_string(len)), ref_stack.back())); } if (len != detail::unknown_size()) @@ -11384,7 +11384,7 @@ class json_sax_dom_callback_parser // check object limit if (JSON_HEDLEY_UNLIKELY(len != detail::unknown_size() && len > ref_stack.back()->max_size())) { - JSON_THROW(out_of_range::create(408, concat("excessive object size: ", std::to_string(len)), ref_stack.back())); + return parse_error(0, "", out_of_range::create(408, concat("excessive object size: ", std::to_string(len)), ref_stack.back())); } } return true; @@ -11503,7 +11503,7 @@ class json_sax_dom_callback_parser // check array limit if (JSON_HEDLEY_UNLIKELY(len != detail::unknown_size() && len > ref_stack.back()->max_size())) { - JSON_THROW(out_of_range::create(408, concat("excessive array size: ", std::to_string(len)), ref_stack.back())); + return parse_error(0, "", out_of_range::create(408, concat("excessive array size: ", std::to_string(len)), ref_stack.back())); } if (len != detail::unknown_size()) diff --git a/tests/src/unit-regression2.cpp b/tests/src/unit-regression2.cpp index 2ae5666ff..2c0cf6549 100644 --- a/tests/src/unit-regression2.cpp +++ b/tests/src/unit-regression2.cpp @@ -834,4 +834,35 @@ TEST_CASE("regression test - parser callback must not lose a duplicate key's pri } } +TEST_CASE("regression test - excessive binary container size honors allow_exceptions=false") +{ + // CBOR array with declared length 2^63 + const std::vector cbor = {0x9b, 0x80, 0, 0, 0, 0, 0, 0, 0}; + // CBOR map with declared length 2^63 + const std::vector cbor_m = {0xbb, 0x80, 0, 0, 0, 0, 0, 0, 0}; + // UBJSON array with declared length 2^63-1 + const std::vector ubj = {'[', '#', 'L', 0x7f, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff}; + // BJData array with declared length 2^63-1 (little endian) + const std::vector bjd = {'[', '#', 'L', 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0x7f}; + + // allow_exceptions=false must report failure instead of throwing/aborting + CHECK(json::from_cbor(cbor, true, false).is_discarded()); + CHECK(json::from_cbor(cbor_m, true, false).is_discarded()); + CHECK(json::from_ubjson(ubj, true, false).is_discarded()); + CHECK(json::from_bjdata(bjd, true, false).is_discarded()); + + // allow_exceptions=true (the default) must still throw exactly as before. + // The exact message text is not checked here: on platforms where + // std::size_t is 32-bit, the CBOR reader's own length-narrowing check + // (get_cbor_container_size(), unrelated to this fix) intercepts a + // declared length of 2^63 before it ever reaches the check this test + // targets, with different (but equally valid, and already correct) + // wording -- see unit-cbor.cpp for coverage of that message. + json _; + CHECK_THROWS_AS(_ = json::from_cbor(cbor), json::out_of_range); + + // regression guard: a genuinely truncated CBOR input must remain discarded + CHECK(json::from_cbor(std::vector {0x9b, 0, 0, 0, 0, 0, 0, 0, 0x02}, true, false).is_discarded()); +} + DOCTEST_CLANG_SUPPRESS_WARNING_POP From d2c1a6a272d28965a066b9a19bbf7662a55b54d1 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Tue, 22 Sep 2026 21:52:31 +0200 Subject: [PATCH 15/64] Reject array insert(pos, first, last) iterators not pointing into an array (#5468) The array-range insert() overload checked that pos fits the current value and that first/last share the same owning value, but never verified that value is itself an array. Passing iterators from an object, a primitive, or null handed value-initialized (singular) std::vector iterators straight to array_t::insert(), which is undefined behavior. Add the missing is_array() check, mirroring the equivalent check already present in the object-range insert() overload. Signed-off-by: Niels Lohmann --- docs/mkdocs/docs/api/basic_json/insert.md | 2 ++ include/nlohmann/json.hpp | 6 ++++++ single_include/nlohmann/json.hpp | 6 ++++++ tests/src/unit-modifiers.cpp | 14 ++++++++++++++ 4 files changed, 28 insertions(+) diff --git a/docs/mkdocs/docs/api/basic_json/insert.md b/docs/mkdocs/docs/api/basic_json/insert.md index 14d5823c1..fcb1e6e44 100644 --- a/docs/mkdocs/docs/api/basic_json/insert.md +++ b/docs/mkdocs/docs/api/basic_json/insert.md @@ -88,6 +88,8 @@ Strong exception safety: if an exception occurs, the original value stays intact do not belong to the same JSON value; example: `"iterators do not fit"` - Throws [`invalid_iterator.211`](../../home/exceptions.md#jsonexceptioninvalid_iterator211) if `first` or `last` are iterators into container for which insert is called; example: `"passed iterators may not belong to container"` + - Throws [`invalid_iterator.202`](../../home/exceptions.md#jsonexceptioninvalid_iterator202) if `first` or `last` + do not point to an array; example: `"iterators first and last must point to arrays"` 4. The function can throw the following exceptions: - Throws [`type_error.309`](../../home/exceptions.md#jsonexceptiontype_error309) if called on JSON values other than arrays; example: `"cannot use insert() with string"` diff --git a/include/nlohmann/json.hpp b/include/nlohmann/json.hpp index 0294a268d..d55efacc2 100644 --- a/include/nlohmann/json.hpp +++ b/include/nlohmann/json.hpp @@ -3511,6 +3511,12 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_THROW(invalid_iterator::create(211, "passed iterators may not belong to container", this)); } + // passed iterators must belong to arrays + if (JSON_HEDLEY_UNLIKELY(!first.m_object->is_array())) + { + JSON_THROW(invalid_iterator::create(202, "iterators first and last must point to arrays", this)); + } + // insert to array and return iterator return insert_iterator(pos, first.m_it.array_iterator, last.m_it.array_iterator); } diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 9b61ac0db..2c73944c0 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -27516,6 +27516,12 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_THROW(invalid_iterator::create(211, "passed iterators may not belong to container", this)); } + // passed iterators must belong to arrays + if (JSON_HEDLEY_UNLIKELY(!first.m_object->is_array())) + { + JSON_THROW(invalid_iterator::create(202, "iterators first and last must point to arrays", this)); + } + // insert to array and return iterator return insert_iterator(pos, first.m_it.array_iterator, last.m_it.array_iterator); } diff --git a/tests/src/unit-modifiers.cpp b/tests/src/unit-modifiers.cpp index de14b3f70..369162772 100644 --- a/tests/src/unit-modifiers.cpp +++ b/tests/src/unit-modifiers.cpp @@ -641,6 +641,20 @@ TEST_CASE("modifiers") CHECK_THROWS_WITH_AS(j_array.insert(j_array.end(), j_other_array.begin(), j_other_array2.end()), "[json.exception.invalid_iterator.210] iterators do not fit", json::invalid_iterator&); } + + SECTION("iterators not pointing into an array") + { + json j_object2 = {{"k", 1}, {"l", 2}}; + json j_primitive = 5; + json j_null; + + CHECK_THROWS_WITH_AS(j_array.insert(j_array.begin(), j_object2.begin(), j_object2.end()), "[json.exception.invalid_iterator.202] iterators first and last must point to arrays", + json::invalid_iterator&); + CHECK_THROWS_WITH_AS(j_array.insert(j_array.begin(), j_primitive.begin(), j_primitive.end()), "[json.exception.invalid_iterator.202] iterators first and last must point to arrays", + json::invalid_iterator&); + CHECK_THROWS_WITH_AS(j_array.insert(j_array.begin(), j_null.begin(), j_null.end()), "[json.exception.invalid_iterator.202] iterators first and last must point to arrays", + json::invalid_iterator&); + } } SECTION("range for object") From 0b20b7e62211be63522b986bd8eef7d7c68f0cf5 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Tue, 22 Sep 2026 21:52:32 +0200 Subject: [PATCH 16/64] Reject MessagePack/BSON binary subtypes that don't fit their wire format (#5469) * Reject MessagePack/BSON binary subtypes that don't fit their wire format Both formats store byte_container_with_subtype's subtype (a uint64_t) in a single byte. The writers cast to std::int8_t/std::uint8_t without a range check, so subtypes above 255 were silently truncated modulo 256 instead of raising an error. Throw out_of_range.413 instead when the subtype exceeds the representable range of 0-255. Signed-off-by: Niels Lohmann * Move the new binary-subtype regression test out of unit-regression2.cpp unit-regression2.cpp is already at the edge of what the MinGW linker can relocate; adding this test's ~26 lines tips test-regression2_cpp20 (clang, Windows) over into "relocation truncated to fit: IMAGE_REL_AMD64_REL32 against `.rdata'" (see 8ce64b9c1 / b82717c8a for the same failure mode). Split the test along format lines instead: MessagePack assertions move to unit-msgpack.cpp, BSON assertions to unit-bson.cpp. The CBOR round-trip guard is dropped as redundant -- unit-cbor.cpp's "Tagged values" section already round-trips subtypes up to 8589934590, far past the 70000 checked here. Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- docs/mkdocs/docs/home/exceptions.md | 15 +++++++++++++++ include/nlohmann/detail/output/binary_writer.hpp | 11 +++++++++++ single_include/nlohmann/json.hpp | 11 +++++++++++ tests/src/unit-bson.cpp | 9 +++++++++ tests/src/unit-msgpack.cpp | 15 +++++++++++++++ 5 files changed, 61 insertions(+) diff --git a/docs/mkdocs/docs/home/exceptions.md b/docs/mkdocs/docs/home/exceptions.md index 09cc8e178..9a7698b2f 100644 --- a/docs/mkdocs/docs/home/exceptions.md +++ b/docs/mkdocs/docs/home/exceptions.md @@ -970,6 +970,21 @@ A JSON Patch `move` operation's `"from"` location is a proper prefix of its `"pa This exception was added in version 3.13.0. Before that, this situation could succeed with a corrupted result: for an array target, removing the "from" element before the "add" step shifted subsequent indices, so "path" silently re-resolved to a different element than intended. +### json.exception.out_of_range.415 + +MessagePack's ext type and BSON's binary subtype are each stored in a single byte. This exception is thrown when serializing a +[`byte_container_with_subtype`](../api/byte_container_with_subtype/index.md) whose subtype exceeds 255. + +!!! failure "Example message" + + ``` + [json.exception.out_of_range.415] subtype 70000 is too large for the MessagePack ext type (max 255) + ``` + +!!! note + + This exception was added in version 3.13.0. Before that, subtypes above 255 were silently truncated modulo 256 instead of raising an error. + ## Further exceptions This exception is thrown in case of errors that cannot be classified with the diff --git a/include/nlohmann/detail/output/binary_writer.hpp b/include/nlohmann/detail/output/binary_writer.hpp index 4bd173257..fe88e7f27 100644 --- a/include/nlohmann/detail/output/binary_writer.hpp +++ b/include/nlohmann/detail/output/binary_writer.hpp @@ -688,6 +688,11 @@ class binary_writer // step 1.5: if this is an ext type, write the subtype if (use_ext) { + if (JSON_HEDLEY_UNLIKELY(j.m_data.m_value.binary->subtype() > (std::numeric_limits::max)())) + { + JSON_THROW(out_of_range::create(415, concat("subtype ", std::to_string(j.m_data.m_value.binary->subtype()), " is too large for the MessagePack ext type (max 255)"), &j)); + } + write_number(static_cast(j.m_data.m_value.binary->subtype())); } @@ -1213,6 +1218,12 @@ class binary_writer write_bson_entry_header(name, 0x05); write_number(to_bson_length(value.size()), true); + + if (value.has_subtype() && JSON_HEDLEY_UNLIKELY(value.subtype() > (std::numeric_limits::max)())) + { + JSON_THROW(out_of_range::create(415, concat("subtype ", std::to_string(value.subtype()), " is too large for the BSON binary subtype (max 255)"), nullptr)); + } + write_number(value.has_subtype() ? static_cast(value.subtype()) : static_cast(0x00)); oa->write_characters(reinterpret_cast(value.data()), value.size()); diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 2c73944c0..f300cbdb1 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -19445,6 +19445,11 @@ class binary_writer // step 1.5: if this is an ext type, write the subtype if (use_ext) { + if (JSON_HEDLEY_UNLIKELY(j.m_data.m_value.binary->subtype() > (std::numeric_limits::max)())) + { + JSON_THROW(out_of_range::create(415, concat("subtype ", std::to_string(j.m_data.m_value.binary->subtype()), " is too large for the MessagePack ext type (max 255)"), &j)); + } + write_number(static_cast(j.m_data.m_value.binary->subtype())); } @@ -19970,6 +19975,12 @@ class binary_writer write_bson_entry_header(name, 0x05); write_number(to_bson_length(value.size()), true); + + if (value.has_subtype() && JSON_HEDLEY_UNLIKELY(value.subtype() > (std::numeric_limits::max)())) + { + JSON_THROW(out_of_range::create(415, concat("subtype ", std::to_string(value.subtype()), " is too large for the BSON binary subtype (max 255)"), nullptr)); + } + write_number(value.has_subtype() ? static_cast(value.subtype()) : static_cast(0x00)); oa->write_characters(reinterpret_cast(value.data()), value.size()); diff --git a/tests/src/unit-bson.cpp b/tests/src/unit-bson.cpp index 153e12d30..669a4bfe1 100644 --- a/tests/src/unit-bson.cpp +++ b/tests/src/unit-bson.cpp @@ -791,6 +791,15 @@ TEST_CASE("BSON") } } +TEST_CASE("regression test - BSON binary subtype rejects a value that doesn't fit a single byte") +{ + json const doc255 = {{"b", json::binary({1, 2}, 255)}}; + CHECK(json::from_bson(json::to_bson(doc255))["b"].get_binary().subtype() == 255); + + CHECK_THROWS_AS(json::to_bson(json{{"b", json::binary({1, 2}, 256)}}), json::out_of_range); + CHECK_THROWS_WITH_AS(json::to_bson(json{{"b", json::binary({1, 2}, 300)}}), "[json.exception.out_of_range.415] subtype 300 is too large for the BSON binary subtype (max 255)", json::out_of_range); +} + TEST_CASE("BSON input/output_adapters") { const json json_representation = diff --git a/tests/src/unit-msgpack.cpp b/tests/src/unit-msgpack.cpp index 74f7f4969..a8892081d 100644 --- a/tests/src/unit-msgpack.cpp +++ b/tests/src/unit-msgpack.cpp @@ -1682,6 +1682,21 @@ TEST_CASE("issue #5405 - array reserve for definite-length MessagePack arrays") } } +TEST_CASE("regression test - MessagePack ext type rejects a subtype that doesn't fit a single byte") +{ + // subtype 0-255 must still round-trip correctly (regression guard, pre-existing behavior) + CHECK(json::from_msgpack(json::to_msgpack(json::binary({1, 2}, 0))).get_binary().subtype() == 0); + CHECK(json::from_msgpack(json::to_msgpack(json::binary({1, 2}, 200))).get_binary().subtype() == 200); + CHECK(json::from_msgpack(json::to_msgpack(json::binary({1, 2}, 255))).get_binary().subtype() == 255); + + // a subtype > 255 must throw instead of silently truncating + CHECK_THROWS_AS(json::to_msgpack(json::binary({1, 2}, 256)), json::out_of_range); + CHECK_THROWS_WITH_AS(json::to_msgpack(json::binary({1, 2}, 70000)), "[json.exception.out_of_range.415] subtype 70000 is too large for the MessagePack ext type (max 255)", json::out_of_range); + + // a binary value with no subtype at all must be unaffected + CHECK(json::from_msgpack(json::to_msgpack(json::binary({1, 2}))).get_binary().has_subtype() == false); +} + // use this testcase outside [hide] to run it with Valgrind TEST_CASE("MessagePack nesting does not consume the call stack") { From f92024b31771395af1e828a757760847e11a26d9 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Wed, 23 Sep 2026 07:48:21 +0200 Subject: [PATCH 17/64] De-duplicate the swap() diagnostic-positions characterization test (#5540) --- tests/src/unit-class_parser.cpp | 95 ++++++--------------------------- 1 file changed, 15 insertions(+), 80 deletions(-) diff --git a/tests/src/unit-class_parser.cpp b/tests/src/unit-class_parser.cpp index e22c4cacf..df4e7270d 100644 --- a/tests/src/unit-class_parser.cpp +++ b/tests/src/unit-class_parser.cpp @@ -2261,86 +2261,6 @@ TEST_CASE("parser class") #endif } -#if JSON_DIAGNOSTIC_POSITIONS - -TEST_CASE("diagnostic positions: value lifetime") -{ - SECTION("copy constructor copies positions, recursively") - { - const std::string s = R"({"a":1,"b":[1,2,3]})"; - const json a = json::parse(s); - const json b = a; // NOLINT(performance-unnecessary-copy-initialization) - - CHECK(b.start_pos() == a.start_pos()); - CHECK(b.end_pos() == a.end_pos()); - CHECK(b["b"].start_pos() == a["b"].start_pos()); - CHECK(b["b"].end_pos() == a["b"].end_pos()); - } - - SECTION("move constructor resets the moved-from value to npos") - { - const std::string s = R"({"a":1,"b":[1,2,3]})"; - json a = json::parse(s); - const auto a_start = a.start_pos(); - const auto a_end = a.end_pos(); - - const json b(std::move(a)); - - CHECK(b.start_pos() == a_start); - CHECK(b.end_pos() == a_end); - - CHECK(a.start_pos() == std::string::npos); // NOLINT(bugprone-use-after-move,clang-analyzer-cplusplus.Move) - CHECK(a.end_pos() == std::string::npos); // NOLINT(bugprone-use-after-move,clang-analyzer-cplusplus.Move) - } - - SECTION("swap() exchanges positions along with the values") - { - // basic_json::swap() (and the friend swap() that forwards to it) used - // to swap only m_data.m_type/m_data.m_value, leaving - // start_position/end_position untouched -- unlike copy-assignment's - // operator=(basic_json), which swaps positions as part of its - // copy-and-swap implementation. After swap(a, b), each value ended up - // with the *other* value's content but its *own* original position. - // This is now fixed so that swap() is consistent with copy-assignment. - json a = json::parse(R"({"a":1})"); - json b = json::parse(R"([1,2,3,4,5])"); - const auto a_start = a.start_pos(); - const auto a_end = a.end_pos(); - const auto b_start = b.start_pos(); - const auto b_end = b.end_pos(); - // lengths (and thus end positions) differ, which is enough to tell - // after the swap whether positions actually moved with the values - CHECK(a_end != b_end); - - using std::swap; - swap(a, b); - - CHECK(a == json::parse(R"([1,2,3,4,5])")); - CHECK(b == json::parse(R"({"a":1})")); - - CHECK(a.start_pos() == b_start); - CHECK(a.end_pos() == b_end); - CHECK(b.start_pos() == a_start); - CHECK(b.end_pos() == a_end); - - // member swap() behaves the same as the free function - json c = json::parse(R"({"a":1})"); - json d = json::parse(R"([1,2,3,4,5])"); - const auto c_start = c.start_pos(); - const auto c_end = c.end_pos(); - const auto d_start = d.start_pos(); - const auto d_end = d.end_pos(); - - c.swap(d); - - CHECK(c.start_pos() == d_start); - CHECK(c.end_pos() == d_end); - CHECK(d.start_pos() == c_start); - CHECK(d.end_pos() == c_end); - } -} -#endif - // this test relies on parse errors being thrown, so it is skipped when // exceptions are disabled (json::parse aborts instead of throwing there) #if !defined(JSON_NOEXCEPTION) @@ -2546,6 +2466,21 @@ TEST_CASE("diagnostic positions: value lifetime, input adapters, and SAX") CHECK(a.end_pos() == b_end); CHECK(b.start_pos() == a_start); CHECK(b.end_pos() == a_end); + + // member swap() behaves the same as the free function + json c = json::parse(R"({"a":1})"); + json d = json::parse(R"([1,2,3,4,5])"); + const auto c_start = c.start_pos(); + const auto c_end = c.end_pos(); + const auto d_start = d.start_pos(); + const auto d_end = d.end_pos(); + + c.swap(d); + + CHECK(c.start_pos() == d_start); + CHECK(c.end_pos() == d_end); + CHECK(d.start_pos() == c_start); + CHECK(d.end_pos() == c_end); } SECTION("mutating a parsed document leaves positions of unrelated values untouched") From 1054b2097e721e8a455285a1b528cc72caf859b7 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Wed, 23 Sep 2026 08:59:46 +0200 Subject: [PATCH 18/64] Speed up binary writing: value-type output sink + byte-swap number encoding (#5286) * Devirtualize binary_writer via a value-type output sink to_cbor/to_msgpack/to_ubjson/to_bjdata/to_bson wrote every byte through output_adapter_t, a shared_ptr whose write_character/write_characters are virtual. Unlike the lexer (templated on a concrete InputAdapterType), the binary writer never got that treatment, so binary output paid a vtable lookup per byte and a make_shared per call. Template binary_writer on an OutputSinkType and give it two concrete, non-virtual sinks: - output_vector_sink: appends straight into a std::vector (push_back / insert), used by the vector-returning to_* convenience functions. No vtable, no shared_ptr; the writes inline. - output_adapter_sink: forwards to a type-erased output_adapter_t, so the existing to_*(j, output_adapter) overloads (streams, strings, custom adapters) keep working exactly as before -- one virtual call each, unchanged. binary_writer keeps a convenience constructor taking output_adapter_t (building the default output_adapter_sink), so the adapter overloads are untouched; only the convenience functions switch to the vector sink. The friend declaration and the basic_json binary_writer alias gain the new (defaulted) template parameter. Output is byte-for-byte identical: verified across ~3000 randomized values plus curated edge cases (all scalar widths, strings with invalid UTF-8, binary, nested arrays/objects) for CBOR, MessagePack, UBJSON (both size/type settings), BJData, and BSON, plus the output_adapter path, in C++11/17/20. Warning-clean under clang -Weverything and the gcc pedantic set; clang-tidy clean on the changed headers; make check-amalgamation clean. Throughput (g++ -O3, vs develop): scalar-dense binary output such as integer arrays ~1.4x; many small to_cbor calls ~1.04x (DOM traversal bound); string/blob-heavy output unchanged (already bulk-bound). No workload regressed. Co-Authored-By: Claude Opus 4.8 Claude-Session: https://claude.ai/code/session_01XAYM1qhSA2FDaDcGfPW3fG Signed-off-by: Niels Lohmann * Fix CI failures from binary_writer output-sink change Four CI jobs failed on the initial commit; all are addressed here without changing any output (binary encodings remain byte-for-byte identical to develop across the differential corpus): 1. ci_test_gcc / cuda (-Werror=duplicated-branches): for number_float_t == float, static_cast(n) is the identity, so write_compact_float's two branches are intentionally identical. Once the concrete vector sink is inlined, GCC constant-folds and diagnoses this (the type-erased path hid it behind a non-inlined virtual call). Silence -Wduplicated-branches for GCC (clang has no such warning) alongside the existing -Wfloat-equal pragma. 2. ci_static_analysis_clang (UBSan nonnull-attribute): binary_writer passes a null pointer with length 0 for empty strings/binary. output_vector_sink / output_adapter_sink declared write_characters JSON_HEDLEY_NON_NULL, so the sanitizer flagged the (harmless) zero-length call once the sink was called directly rather than through the attribute-free virtual base. Drop the attribute from both sinks, matching the pre-existing behavior. 3. ci_cpplint (build/include_what_you_use): output_adapter_sink uses std::move; add #include . 4. ci_cuda_example (nvcc 11.8): NVCC's front end rejects the default template argument on the binary_writer alias template. Revert the alias to its original single-parameter form (relying on binary_writer's own defaulted OutputSinkType) and spell out the full type in the vector-sink convenience functions. Co-Authored-By: Claude Opus 4.8 Claude-Session: https://claude.ai/code/session_01XAYM1qhSA2FDaDcGfPW3fG Signed-off-by: Niels Lohmann * Encode big-endian numbers with a byte swap instead of std::reverse write_number() reordered multi-byte numbers for the big-endian formats (CBOR/MessagePack/UBJSON) with std::reverse over the byte array. GCC lowered only some sizes to a bswap; clang kept a scalar byte shuffle (0 bswap instructions in the CBOR number path). Replace the reverse with size-dispatched __builtin_bswap16/32/64 helpers (portable shift fallback for other compilers; std::reverse retained for exotic sizes such as a long double number_float_t). Codegen: the CBOR number path now emits bswap on both compilers (gcc 2 -> 16, clang 0 -> 4). Output is byte-for-byte identical to the previous implementation across the binary differential corpus. Throughput (isolated vs the std::reverse version, best of 9): CBOR int64 array gcc +7% clang +10% CBOR uint16 array gcc +27% clang flat Modest but consistent on number-dense encodings; negligible on string/blob-heavy output, as expected. Co-Authored-By: Claude Opus 4.8 Claude-Session: https://claude.ai/code/session_01XAYM1qhSA2FDaDcGfPW3fG Signed-off-by: Niels Lohmann * Reserve output capacity up front for binary serialization The vector-returning to_cbor/to_msgpack/to_ubjson/to_bjdata/to_bson grew the output buffer purely by geometric reallocation. Reserving an estimate up front avoids the early reallocations, which is the dominant per-byte cost for array/object-heavy output. The estimate (binary_reserve_hint) is deliberately conservative and safe against untrusted input: it consults only the top-level element count (O(1), no walk of the DOM), guards the multiplication against overflow, and clamps the result to a fixed 1 MiB ceiling, so a large or hostile DOM can never force an oversized allocation here. The buffer still grows geometrically past the hint, so an underestimate only costs a few later reallocations; scalars/strings/binary are written in one shot and get no hint. Reserving capacity does not change the bytes produced. Throughput (g++/clang -O3, vs the previous commit): cbor int array +10% / +13% cbor object array +20% / +38% Output is byte-for-byte identical to develop across the binary differential corpus. Co-Authored-By: Claude Opus 4.8 Claude-Session: https://claude.ai/code/session_01XAYM1qhSA2FDaDcGfPW3fG Signed-off-by: Niels Lohmann * Address review findings on the binary writer output sinks - binary_reserve_hint(): the 4-bytes-per-element estimate over-reserved by up to 4x for arrays of small scalars (CBOR encodes 0..23 in one byte), and the returned vector kept that capacity. Make the hint a strict lower bound on the encoded size instead, which also removes the 1 MiB clamp whose branch no test could reach (the largest container in the suite has 65793 elements). - Guard the -Wduplicated-branches pragma with __GNUC__ >= 7. The warning does not exist before GCC 7, so naming it made GCC 4.8/4.9/5/6 - which the CI matrix still builds - warn under -Wpragmas on every including translation unit, breaking downstream -Werror builds. - Constrain the adapter constructor of binary_writer with the enable_if its documentation already claimed, so a writer over some other sink type is no longer advertised as constructible from an output adapter. - Let output_vector_adapter wrap output_vector_sink rather than duplicating the append logic, so the type-erased and templated paths share one implementation. - Collapse the three copies of the memcpy/byte_swap/memcpy dance into a single byte_swap_buffer() helper, and add the MSVC _byteswap_* intrinsics so MSVC no longer falls back to the scalar shuffle this change exists to eliminate. - Add a vector_writer() helper for the five vector-returning to_* overloads instead of spelling out the writer type at each call site, and drop a dead default member initializer on output_adapter_sink. - New tests: the vector sink and the adapter sink must produce identical bytes for every format (the two to_* overloads no longer delegate to each other and could otherwise drift), and binary_reserve_hint() must never exceed the size actually written. Signed-off-by: Niels Lohmann * Route the -Wduplicated-branches pragma through Hedley Match #5485, which moved the binary writer's hand-rolled diagnostic pragmas onto JSON_HEDLEY_PRAGMA (merged into develop while this branch was open). The devirtualization's -Wduplicated-branches suppression in write_compact_float was the one raw '#pragma GCC diagnostic' left; it now uses JSON_HEDLEY_PRAGMA like the adjacent -Wfloat-equal line, still guarded to GCC >= 7 and non-clang (the warning exists only there). Co-Authored-By: Claude Claude-Session: https://claude.ai/code/session_01XAYM1qhSA2FDaDcGfPW3fG Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann Co-authored-by: Claude Opus 4.8 --- .../nlohmann/detail/output/binary_writer.hpp | 458 +++++++++----- .../detail/output/output_adapters.hpp | 84 ++- include/nlohmann/json.hpp | 25 +- single_include/nlohmann/json.hpp | 567 ++++++++++++------ tests/src/unit-binary_writer_sinks.cpp | 198 ++++++ 5 files changed, 992 insertions(+), 340 deletions(-) create mode 100644 tests/src/unit-binary_writer_sinks.cpp diff --git a/include/nlohmann/detail/output/binary_writer.hpp b/include/nlohmann/detail/output/binary_writer.hpp index fe88e7f27..a355f1c15 100644 --- a/include/nlohmann/detail/output/binary_writer.hpp +++ b/include/nlohmann/detail/output/binary_writer.hpp @@ -16,9 +16,14 @@ #include // memcpy #include // numeric_limits #include // string +#include // enable_if, is_constructible #include // move #include // vector +#ifdef _MSC_VER + #include // _byteswap_ushort, _byteswap_ulong, _byteswap_uint64 +#endif + #include #include #include @@ -39,10 +44,41 @@ enum class bjdata_version_t // binary writer // /////////////////// +/*! +@brief capacity hint for binary serialization into a std::vector + +Returns a *lower* bound on the number of bytes the serialization will produce, +so that writing an array/object of many elements does not start reallocating +from an empty buffer. Every array element occupies at least one byte in every +supported binary format, and every object entry at least two (a key of at least +one byte plus a value of at least one), plus one byte for the container header, +so the hint can never exceed the final size and the returned vector is never +left holding capacity the caller did not ask for. The buffer still grows +geometrically past the hint, so under-reserving only costs a few later +reallocations. Only the top-level element count is consulted (O(1), no walk of +the DOM); a single scalar, string, or binary value is written in one shot and +needs no hint. +*/ +template +std::size_t binary_reserve_hint(const BasicJsonType& j) +{ + if (j.is_array()) + { + return j.size() + 1; + } + + if (j.is_object()) + { + return (j.size() * 2) + 1; + } + + return 0; +} + /*! @brief serialization to CBOR and MessagePack values */ -template +template> class binary_writer { using string_t = typename BasicJsonType::string_t; @@ -53,12 +89,28 @@ class binary_writer /*! @brief create a binary writer + @param[in] sink output sink to write to (a value-type sink such as + output_vector_sink, or output_adapter_sink wrapping a + type-erased output adapter) + */ + explicit binary_writer(OutputSinkType sink) : oa(std::move(sink)) + {} + + /*! + @brief create a binary writer from a type-erased output adapter + + Convenience constructor for the default (output_adapter_sink) sink so the + `output_adapter`-based overloads keep constructing the writer directly from + an adapter. Constrained to sinks that can actually be built from an adapter, + so that a writer over some other sink type is not advertised as constructible + from one. + @param[in] adapter output adapter to write to */ - explicit binary_writer(output_adapter_t adapter) : oa(std::move(adapter)) - { - JSON_ASSERT(oa); - } + template < typename SinkType = OutputSinkType, + typename std::enable_if < std::is_constructible>::value, int >::type = 0 > + explicit binary_writer(output_adapter_t adapter) : oa(SinkType(std::move(adapter))) + {} /*! @param[in] j JSON value to serialize @@ -99,15 +151,15 @@ class binary_writer { case value_t::null: { - oa->write_character(to_char_type(0xF6)); + oa.write_character(to_char_type(0xF6)); break; } case value_t::boolean: { - oa->write_character(j.m_data.m_value.boolean - ? to_char_type(0xF5) - : to_char_type(0xF4)); + oa.write_character(j.m_data.m_value.boolean + ? to_char_type(0xF5) + : to_char_type(0xF4)); break; } @@ -124,22 +176,22 @@ class binary_writer } else if (j.m_data.m_value.number_integer <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x18)); + oa.write_character(to_char_type(0x18)); write_number(static_cast(j.m_data.m_value.number_integer)); } else if (j.m_data.m_value.number_integer <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x19)); + oa.write_character(to_char_type(0x19)); write_number(static_cast(j.m_data.m_value.number_integer)); } else if (j.m_data.m_value.number_integer <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x1A)); + oa.write_character(to_char_type(0x1A)); write_number(static_cast(j.m_data.m_value.number_integer)); } else { - oa->write_character(to_char_type(0x1B)); + oa.write_character(to_char_type(0x1B)); write_number(static_cast(j.m_data.m_value.number_integer)); } } @@ -154,22 +206,22 @@ class binary_writer } else if (positive_number <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x38)); + oa.write_character(to_char_type(0x38)); write_number(static_cast(positive_number)); } else if (positive_number <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x39)); + oa.write_character(to_char_type(0x39)); write_number(static_cast(positive_number)); } else if (positive_number <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x3A)); + oa.write_character(to_char_type(0x3A)); write_number(static_cast(positive_number)); } else { - oa->write_character(to_char_type(0x3B)); + oa.write_character(to_char_type(0x3B)); write_number(static_cast(positive_number)); } } @@ -184,22 +236,22 @@ class binary_writer } else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x18)); + oa.write_character(to_char_type(0x18)); write_number(static_cast(j.m_data.m_value.number_unsigned)); } else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x19)); + oa.write_character(to_char_type(0x19)); write_number(static_cast(j.m_data.m_value.number_unsigned)); } else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x1A)); + oa.write_character(to_char_type(0x1A)); write_number(static_cast(j.m_data.m_value.number_unsigned)); } else { - oa->write_character(to_char_type(0x1B)); + oa.write_character(to_char_type(0x1B)); write_number(static_cast(j.m_data.m_value.number_unsigned)); } break; @@ -210,16 +262,16 @@ class binary_writer if (std::isnan(j.m_data.m_value.number_float)) { // NaN is 0xf97e00 in CBOR - oa->write_character(to_char_type(0xF9)); - oa->write_character(to_char_type(0x7E)); - oa->write_character(to_char_type(0x00)); + oa.write_character(to_char_type(0xF9)); + oa.write_character(to_char_type(0x7E)); + oa.write_character(to_char_type(0x00)); } else if (std::isinf(j.m_data.m_value.number_float)) { // Infinity is 0xf97c00, -Infinity is 0xf9fc00 - oa->write_character(to_char_type(0xf9)); - oa->write_character(j.m_data.m_value.number_float > 0 ? to_char_type(0x7C) : to_char_type(0xFC)); - oa->write_character(to_char_type(0x00)); + oa.write_character(to_char_type(0xf9)); + oa.write_character(j.m_data.m_value.number_float > 0 ? to_char_type(0x7C) : to_char_type(0xFC)); + oa.write_character(to_char_type(0x00)); } else { @@ -238,31 +290,31 @@ class binary_writer } else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x78)); + oa.write_character(to_char_type(0x78)); write_number(static_cast(N)); } else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x79)); + oa.write_character(to_char_type(0x79)); write_number(static_cast(N)); } else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x7A)); + oa.write_character(to_char_type(0x7A)); write_number(static_cast(N)); } // LCOV_EXCL_START else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x7B)); + oa.write_character(to_char_type(0x7B)); write_number(static_cast(N)); } // LCOV_EXCL_STOP // step 2: write the string - oa->write_characters( - reinterpret_cast(j.m_data.m_value.string->data()), - j.m_data.m_value.string->size()); + oa.write_characters( + reinterpret_cast(j.m_data.m_value.string->data()), + j.m_data.m_value.string->size()); break; } @@ -276,23 +328,23 @@ class binary_writer } else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x98)); + oa.write_character(to_char_type(0x98)); write_number(static_cast(N)); } else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x99)); + oa.write_character(to_char_type(0x99)); write_number(static_cast(N)); } else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x9A)); + oa.write_character(to_char_type(0x9A)); write_number(static_cast(N)); } // LCOV_EXCL_START else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x9B)); + oa.write_character(to_char_type(0x9B)); write_number(static_cast(N)); } // LCOV_EXCL_STOP @@ -339,31 +391,31 @@ class binary_writer } else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x58)); + oa.write_character(to_char_type(0x58)); write_number(static_cast(N)); } else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x59)); + oa.write_character(to_char_type(0x59)); write_number(static_cast(N)); } else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x5A)); + oa.write_character(to_char_type(0x5A)); write_number(static_cast(N)); } // LCOV_EXCL_START else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x5B)); + oa.write_character(to_char_type(0x5B)); write_number(static_cast(N)); } // LCOV_EXCL_STOP // step 2: write each element - oa->write_characters( - reinterpret_cast(j.m_data.m_value.binary->data()), - N); + oa.write_characters( + reinterpret_cast(j.m_data.m_value.binary->data()), + N); break; } @@ -378,23 +430,23 @@ class binary_writer } else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0xB8)); + oa.write_character(to_char_type(0xB8)); write_number(static_cast(N)); } else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0xB9)); + oa.write_character(to_char_type(0xB9)); write_number(static_cast(N)); } else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0xBA)); + oa.write_character(to_char_type(0xBA)); write_number(static_cast(N)); } // LCOV_EXCL_START else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0xBB)); + oa.write_character(to_char_type(0xBB)); write_number(static_cast(N)); } // LCOV_EXCL_STOP @@ -423,15 +475,15 @@ class binary_writer { case value_t::null: // nil { - oa->write_character(to_char_type(0xC0)); + oa.write_character(to_char_type(0xC0)); break; } case value_t::boolean: // true and false { - oa->write_character(j.m_data.m_value.boolean - ? to_char_type(0xC3) - : to_char_type(0xC2)); + oa.write_character(j.m_data.m_value.boolean + ? to_char_type(0xC3) + : to_char_type(0xC2)); break; } @@ -450,25 +502,25 @@ class binary_writer else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) { // uint 8 - oa->write_character(to_char_type(0xCC)); + oa.write_character(to_char_type(0xCC)); write_number(static_cast(j.m_data.m_value.number_integer)); } else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) { // uint 16 - oa->write_character(to_char_type(0xCD)); + oa.write_character(to_char_type(0xCD)); write_number(static_cast(j.m_data.m_value.number_integer)); } else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) { // uint 32 - oa->write_character(to_char_type(0xCE)); + oa.write_character(to_char_type(0xCE)); write_number(static_cast(j.m_data.m_value.number_integer)); } else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) { // uint 64 - oa->write_character(to_char_type(0xCF)); + oa.write_character(to_char_type(0xCF)); write_number(static_cast(j.m_data.m_value.number_integer)); } } @@ -483,28 +535,28 @@ class binary_writer j.m_data.m_value.number_integer <= (std::numeric_limits::max)()) { // int 8 - oa->write_character(to_char_type(0xD0)); + oa.write_character(to_char_type(0xD0)); write_number(static_cast(j.m_data.m_value.number_integer)); } else if (j.m_data.m_value.number_integer >= (std::numeric_limits::min)() && j.m_data.m_value.number_integer <= (std::numeric_limits::max)()) { // int 16 - oa->write_character(to_char_type(0xD1)); + oa.write_character(to_char_type(0xD1)); write_number(static_cast(j.m_data.m_value.number_integer)); } else if (j.m_data.m_value.number_integer >= (std::numeric_limits::min)() && j.m_data.m_value.number_integer <= (std::numeric_limits::max)()) { // int 32 - oa->write_character(to_char_type(0xD2)); + oa.write_character(to_char_type(0xD2)); write_number(static_cast(j.m_data.m_value.number_integer)); } else if (j.m_data.m_value.number_integer >= (std::numeric_limits::min)() && j.m_data.m_value.number_integer <= (std::numeric_limits::max)()) { // int 64 - oa->write_character(to_char_type(0xD3)); + oa.write_character(to_char_type(0xD3)); write_number(static_cast(j.m_data.m_value.number_integer)); } } @@ -521,25 +573,25 @@ class binary_writer else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) { // uint 8 - oa->write_character(to_char_type(0xCC)); + oa.write_character(to_char_type(0xCC)); write_number(static_cast(j.m_data.m_value.number_integer)); } else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) { // uint 16 - oa->write_character(to_char_type(0xCD)); + oa.write_character(to_char_type(0xCD)); write_number(static_cast(j.m_data.m_value.number_integer)); } else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) { // uint 32 - oa->write_character(to_char_type(0xCE)); + oa.write_character(to_char_type(0xCE)); write_number(static_cast(j.m_data.m_value.number_integer)); } else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) { // uint 64 - oa->write_character(to_char_type(0xCF)); + oa.write_character(to_char_type(0xCF)); write_number(static_cast(j.m_data.m_value.number_integer)); } break; @@ -563,26 +615,26 @@ class binary_writer else if (N <= (std::numeric_limits::max)()) { // str 8 - oa->write_character(to_char_type(0xD9)); + oa.write_character(to_char_type(0xD9)); write_number(static_cast(N)); } else if (N <= (std::numeric_limits::max)()) { // str 16 - oa->write_character(to_char_type(0xDA)); + oa.write_character(to_char_type(0xDA)); write_number(static_cast(N)); } else if (N <= (std::numeric_limits::max)()) { // str 32 - oa->write_character(to_char_type(0xDB)); + oa.write_character(to_char_type(0xDB)); write_number(static_cast(N)); } // step 2: write the string - oa->write_characters( - reinterpret_cast(j.m_data.m_value.string->data()), - j.m_data.m_value.string->size()); + oa.write_characters( + reinterpret_cast(j.m_data.m_value.string->data()), + j.m_data.m_value.string->size()); break; } @@ -598,13 +650,13 @@ class binary_writer else if (N <= (std::numeric_limits::max)()) { // array 16 - oa->write_character(to_char_type(0xDC)); + oa.write_character(to_char_type(0xDC)); write_number(static_cast(N)); } else if (N <= (std::numeric_limits::max)()) { // array 32 - oa->write_character(to_char_type(0xDD)); + oa.write_character(to_char_type(0xDD)); write_number(static_cast(N)); } @@ -660,7 +712,7 @@ class binary_writer fixed = false; } - oa->write_character(to_char_type(output_type)); + oa.write_character(to_char_type(output_type)); if (!fixed) { write_number(static_cast(N)); @@ -672,7 +724,7 @@ class binary_writer ? 0xC8 // ext 16 : 0xC5; // bin 16 - oa->write_character(to_char_type(output_type)); + oa.write_character(to_char_type(output_type)); write_number(static_cast(N)); } else if (N <= (std::numeric_limits::max)()) @@ -681,7 +733,7 @@ class binary_writer ? 0xC9 // ext 32 : 0xC6; // bin 32 - oa->write_character(to_char_type(output_type)); + oa.write_character(to_char_type(output_type)); write_number(static_cast(N)); } @@ -697,9 +749,9 @@ class binary_writer } // step 2: write the byte string - oa->write_characters( - reinterpret_cast(j.m_data.m_value.binary->data()), - N); + oa.write_characters( + reinterpret_cast(j.m_data.m_value.binary->data()), + N); break; } @@ -716,13 +768,13 @@ class binary_writer else if (N <= (std::numeric_limits::max)()) { // map 16 - oa->write_character(to_char_type(0xDE)); + oa.write_character(to_char_type(0xDE)); write_number(static_cast(N)); } else if (N <= (std::numeric_limits::max)()) { // map 32 - oa->write_character(to_char_type(0xDF)); + oa.write_character(to_char_type(0xDF)); write_number(static_cast(N)); } @@ -761,7 +813,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('Z')); + oa.write_character(to_char_type('Z')); } break; } @@ -770,9 +822,9 @@ class binary_writer { if (add_prefix) { - oa->write_character(j.m_data.m_value.boolean - ? to_char_type('T') - : to_char_type('F')); + oa.write_character(j.m_data.m_value.boolean + ? to_char_type('T') + : to_char_type('F')); } break; } @@ -799,12 +851,12 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('S')); + oa.write_character(to_char_type('S')); } write_number_with_ubjson_prefix(j.m_data.m_value.string->size(), true, use_bjdata); - oa->write_characters( - reinterpret_cast(j.m_data.m_value.string->data()), - j.m_data.m_value.string->size()); + oa.write_characters( + reinterpret_cast(j.m_data.m_value.string->data()), + j.m_data.m_value.string->size()); break; } @@ -812,7 +864,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('[')); + oa.write_character(to_char_type('[')); } bool prefix_required = true; @@ -844,14 +896,14 @@ class binary_writer && !(use_bjdata && std::find(bjdx.begin(), bjdx.end(), first_prefix) != bjdx.end())) { prefix_required = false; - oa->write_character(to_char_type('$')); - oa->write_character(first_prefix); + oa.write_character(to_char_type('$')); + oa.write_character(first_prefix); } } if (use_count) { - oa->write_character(to_char_type('#')); + oa.write_character(to_char_type('#')); write_number_with_ubjson_prefix(j.m_data.m_value.array->size(), true, use_bjdata); } @@ -862,7 +914,7 @@ class binary_writer if (!use_count) { - oa->write_character(to_char_type(']')); + oa.write_character(to_char_type(']')); } break; @@ -872,7 +924,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('[')); + oa.write_character(to_char_type('[')); } if (use_type && (bjdata_draft3 || !j.m_data.m_value.binary->empty())) @@ -881,36 +933,36 @@ class binary_writer { JSON_THROW(other_error::create(502, "use_type requires use_size = true", &j)); } - oa->write_character(to_char_type('$')); - oa->write_character(bjdata_draft3 ? 'B' : 'U'); + oa.write_character(to_char_type('$')); + oa.write_character(bjdata_draft3 ? 'B' : 'U'); } if (use_count) { - oa->write_character(to_char_type('#')); + oa.write_character(to_char_type('#')); write_number_with_ubjson_prefix(j.m_data.m_value.binary->size(), true, use_bjdata); } if (use_type) { - oa->write_characters( - reinterpret_cast(j.m_data.m_value.binary->data()), - j.m_data.m_value.binary->size()); + oa.write_characters( + reinterpret_cast(j.m_data.m_value.binary->data()), + j.m_data.m_value.binary->size()); } else { for (size_t i = 0; i < j.m_data.m_value.binary->size(); ++i) { - oa->write_character(to_char_type(bjdata_draft3 ? 'B' : 'U')); + oa.write_character(to_char_type(bjdata_draft3 ? 'B' : 'U')); // the cast is needed for binary types whose value type // is not an integer (e.g., std::byte) - oa->write_character(to_char_type(static_cast(j.m_data.m_value.binary->data()[i]))); + oa.write_character(to_char_type(static_cast(j.m_data.m_value.binary->data()[i]))); } } if (!use_count) { - oa->write_character(to_char_type(']')); + oa.write_character(to_char_type(']')); } break; @@ -928,7 +980,7 @@ class binary_writer if (add_prefix) { - oa->write_character(to_char_type('{')); + oa.write_character(to_char_type('{')); } bool prefix_required = true; @@ -950,29 +1002,29 @@ class binary_writer if (same_prefix && !(use_bjdata && std::find(bjdx.begin(), bjdx.end(), first_prefix) != bjdx.end())) { prefix_required = false; - oa->write_character(to_char_type('$')); - oa->write_character(first_prefix); + oa.write_character(to_char_type('$')); + oa.write_character(first_prefix); } } if (use_count) { - oa->write_character(to_char_type('#')); + oa.write_character(to_char_type('#')); write_number_with_ubjson_prefix(j.m_data.m_value.object->size(), true, use_bjdata); } for (const auto& el : *j.m_data.m_value.object) { write_number_with_ubjson_prefix(el.first.size(), true, use_bjdata); - oa->write_characters( - reinterpret_cast(el.first.data()), - el.first.size()); + oa.write_characters( + reinterpret_cast(el.first.data()), + el.first.size()); write_ubjson(el.second, use_count, use_type, prefix_required, use_bjdata, bjdata_version); } if (!use_count) { - oa->write_character(to_char_type('}')); + oa.write_character(to_char_type('}')); } break; @@ -1026,13 +1078,13 @@ class binary_writer void write_bson_entry_header(const string_t& name, const std::uint8_t element_type) { - oa->write_character(to_char_type(element_type)); - oa->write_characters( - reinterpret_cast(name.data()), - name.size()); + oa.write_character(to_char_type(element_type)); + oa.write_characters( + reinterpret_cast(name.data()), + name.size()); // the terminating null byte is written explicitly rather than taken // from the buffer, so that string_t::data() need not be null-terminated - oa->write_character(to_char_type(0x00)); + oa.write_character(to_char_type(0x00)); } /*! @@ -1042,7 +1094,7 @@ class binary_writer const bool value) { write_bson_entry_header(name, 0x08); - oa->write_character(value ? to_char_type(0x01) : to_char_type(0x00)); + oa.write_character(value ? to_char_type(0x01) : to_char_type(0x00)); } /*! @@ -1072,12 +1124,12 @@ class binary_writer write_bson_entry_header(name, 0x02); write_number(to_bson_length(value.size() + 1ul), true); - oa->write_characters( - reinterpret_cast(value.data()), - value.size()); + oa.write_characters( + reinterpret_cast(value.data()), + value.size()); // the terminating null byte is written explicitly rather than taken // from the buffer, so that string_t::data() need not be null-terminated - oa->write_character(to_char_type(0x00)); + oa.write_character(to_char_type(0x00)); } /*! @@ -1206,7 +1258,7 @@ class binary_writer write_bson_element(string_t(key.data(), key.size()), el); } - oa->write_character(to_char_type(0x00)); + oa.write_character(to_char_type(0x00)); } /*! @@ -1226,7 +1278,7 @@ class binary_writer write_number(value.has_subtype() ? static_cast(value.subtype()) : static_cast(0x00)); - oa->write_characters(reinterpret_cast(value.data()), value.size()); + oa.write_characters(reinterpret_cast(value.data()), value.size()); } /*! @@ -1352,7 +1404,7 @@ class binary_writer write_bson_element(el.first, el.second); } - oa->write_character(to_char_type(0x00)); + oa.write_character(to_char_type(0x00)); } ////////// @@ -1396,7 +1448,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(get_ubjson_float_prefix(n)); + oa.write_character(get_ubjson_float_prefix(n)); } write_number(n, use_bjdata); } @@ -1412,7 +1464,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('i')); // int8 + oa.write_character(to_char_type('i')); // int8 } write_number(static_cast(n), use_bjdata); } @@ -1420,7 +1472,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('U')); // uint8 + oa.write_character(to_char_type('U')); // uint8 } write_number(static_cast(n), use_bjdata); } @@ -1428,7 +1480,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('I')); // int16 + oa.write_character(to_char_type('I')); // int16 } write_number(static_cast(n), use_bjdata); } @@ -1436,7 +1488,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('u')); // uint16 - bjdata only + oa.write_character(to_char_type('u')); // uint16 - bjdata only } write_number(static_cast(n), use_bjdata); } @@ -1444,7 +1496,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('l')); // int32 + oa.write_character(to_char_type('l')); // int32 } write_number(static_cast(n), use_bjdata); } @@ -1452,7 +1504,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('m')); // uint32 - bjdata only + oa.write_character(to_char_type('m')); // uint32 - bjdata only } write_number(static_cast(n), use_bjdata); } @@ -1460,7 +1512,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('L')); // int64 + oa.write_character(to_char_type('L')); // int64 } write_number(static_cast(n), use_bjdata); } @@ -1468,7 +1520,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('M')); // uint64 - bjdata only + oa.write_character(to_char_type('M')); // uint64 - bjdata only } write_number(static_cast(n), use_bjdata); } @@ -1476,14 +1528,14 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('H')); // high-precision number + oa.write_character(to_char_type('H')); // high-precision number } const auto number = BasicJsonType(n).dump(); write_number_with_ubjson_prefix(number.size(), true, use_bjdata); for (std::size_t i = 0; i < number.size(); ++i) { - oa->write_character(to_char_type(static_cast(number[i]))); + oa.write_character(to_char_type(static_cast(number[i]))); } } } @@ -1500,7 +1552,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('i')); // int8 + oa.write_character(to_char_type('i')); // int8 } write_number(static_cast(n), use_bjdata); } @@ -1508,7 +1560,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('U')); // uint8 + oa.write_character(to_char_type('U')); // uint8 } write_number(static_cast(n), use_bjdata); } @@ -1516,7 +1568,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('I')); // int16 + oa.write_character(to_char_type('I')); // int16 } write_number(static_cast(n), use_bjdata); } @@ -1524,7 +1576,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('u')); // uint16 - bjdata only + oa.write_character(to_char_type('u')); // uint16 - bjdata only } write_number(static_cast(n), use_bjdata); } @@ -1532,7 +1584,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('l')); // int32 + oa.write_character(to_char_type('l')); // int32 } write_number(static_cast(n), use_bjdata); } @@ -1540,7 +1592,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('m')); // uint32 - bjdata only + oa.write_character(to_char_type('m')); // uint32 - bjdata only } write_number(static_cast(n), use_bjdata); } @@ -1548,7 +1600,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('L')); // int64 + oa.write_character(to_char_type('L')); // int64 } write_number(static_cast(n), use_bjdata); } @@ -1557,14 +1609,14 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('H')); // high-precision number + oa.write_character(to_char_type('H')); // high-precision number } const auto number = BasicJsonType(n).dump(); write_number_with_ubjson_prefix(number.size(), true, use_bjdata); for (std::size_t i = 0; i < number.size(); ++i) { - oa->write_character(to_char_type(static_cast(number[i]))); + oa.write_character(to_char_type(static_cast(number[i]))); } } // LCOV_EXCL_STOP @@ -1856,10 +1908,10 @@ class binary_writer } } - oa->write_character('['); - oa->write_character('$'); - oa->write_character(dtype); - oa->write_character('#'); + oa.write_character('['); + oa.write_character('$'); + oa.write_character(dtype); + oa.write_character('#'); key = "_ArraySize_"; write_ubjson(value.at(key), use_count, use_type, true, true, bjdata_version); @@ -1955,6 +2007,87 @@ class binary_writer On the other hand, BSON and BJData use little endian and should reorder on big endian systems. */ + // single-instruction byte swaps (compilers lower these to bswap/rev/movbe); + // used to emit big-endian numbers without a per-byte std::reverse loop + static std::uint16_t byte_swap(std::uint16_t x) noexcept + { +#if defined(__GNUC__) || defined(__clang__) + return __builtin_bswap16(x); +#elif defined(_MSC_VER) + return _byteswap_ushort(x); +#else + return static_cast((x >> 8) | (x << 8)); +#endif + } + + static std::uint32_t byte_swap(std::uint32_t x) noexcept + { +#if defined(__GNUC__) || defined(__clang__) + return __builtin_bswap32(x); +#elif defined(_MSC_VER) + return _byteswap_ulong(x); +#else + return ((x & 0x000000FFu) << 24) | ((x & 0x0000FF00u) << 8) + | ((x & 0x00FF0000u) >> 8) | ((x & 0xFF000000u) >> 24); +#endif + } + + static std::uint64_t byte_swap(std::uint64_t x) noexcept + { +#if defined(__GNUC__) || defined(__clang__) + return __builtin_bswap64(x); +#elif defined(_MSC_VER) + return _byteswap_uint64(x); +#else + x = ((x & 0x00000000FFFFFFFFull) << 32) | ((x & 0xFFFFFFFF00000000ull) >> 32); + x = ((x & 0x0000FFFF0000FFFFull) << 16) | ((x & 0xFFFF0000FFFF0000ull) >> 16); + x = ((x & 0x00FF00FF00FF00FFull) << 8) | ((x & 0xFF00FF00FF00FF00ull) >> 8); + return x; +#endif + } + + /*! + @brief reverse the bytes of a buffer by byte-swapping it as UIntType + + Loading the buffer into an unsigned integer of the same width and swapping + that is what lets the compiler emit a single bswap/rev/movbe; reversing the + buffer element by element does not reliably get there (clang keeps a scalar + shuffle). The two memcpy calls are the only portable way to reinterpret the + bytes and are folded away by every optimizer. + */ + template + static void byte_swap_buffer(std::array& a) noexcept + { + static_assert(sizeof(UIntType) == N, "swap width must match the buffer size"); + UIntType v{}; + std::memcpy(&v, a.data(), sizeof(v)); + v = byte_swap(v); + std::memcpy(a.data(), &v, sizeof(v)); + } + + // reverse the bytes of a fixed-size buffer; a single byte_swap() for the + // common 2/4/8-byte number payloads, std::reverse for any other size + static void reverse_bytes(std::array& a) noexcept + { + byte_swap_buffer(a); + } + + static void reverse_bytes(std::array& a) noexcept + { + byte_swap_buffer(a); + } + + static void reverse_bytes(std::array& a) noexcept + { + byte_swap_buffer(a); + } + + template + static void reverse_bytes(std::array& a) noexcept + { + std::reverse(a.begin(), a.end()); + } + template void write_number(const NumberType n, const bool OutputIsLittleEndian = false) { @@ -1966,10 +2099,10 @@ class binary_writer if (is_little_endian != OutputIsLittleEndian) { // reverse byte order prior to conversion if necessary - std::reverse(vec.begin(), vec.end()); + reverse_bytes(vec); } - oa->write_characters(vec.data(), sizeof(NumberType)); + oa.write_characters(vec.data(), sizeof(NumberType)); } void write_compact_float(const number_float_t n, detail::input_format_t format) @@ -1977,21 +2110,30 @@ class binary_writer #ifdef __GNUC__ JSON_HEDLEY_DIAGNOSTIC_PUSH JSON_HEDLEY_PRAGMA(GCC diagnostic ignored "-Wfloat-equal") +#endif + // When number_float_t is float, static_cast(n) is the identity and + // both branches below are intentionally identical (the "compact" float + // representation is the value itself). Only GCC diagnoses this, and only + // when the sink calls are inlined; clang has no such warning. + // (-Wduplicated-branches only exists from GCC 7 on; naming it on an older + // GCC would itself warn under -Wpragmas) +#if defined(__GNUC__) && !defined(__clang__) && (__GNUC__ >= 7) + JSON_HEDLEY_PRAGMA(GCC diagnostic ignored "-Wduplicated-branches") #endif if (!std::isfinite(n) || ((static_cast(n) >= static_cast(std::numeric_limits::lowest()) && static_cast(n) <= static_cast((std::numeric_limits::max)()) && static_cast(static_cast(n)) == static_cast(n)))) { - oa->write_character(format == detail::input_format_t::cbor - ? get_cbor_float_prefix(static_cast(n)) - : get_msgpack_float_prefix(static_cast(n))); + oa.write_character(format == detail::input_format_t::cbor + ? get_cbor_float_prefix(static_cast(n)) + : get_msgpack_float_prefix(static_cast(n))); write_number(static_cast(n)); } else { - oa->write_character(format == detail::input_format_t::cbor - ? get_cbor_float_prefix(n) - : get_msgpack_float_prefix(n)); + oa.write_character(format == detail::input_format_t::cbor + ? get_cbor_float_prefix(n) + : get_msgpack_float_prefix(n)); write_number(n); } #ifdef __GNUC__ @@ -2058,7 +2200,7 @@ class binary_writer const bool is_little_endian = little_endianness(); /// the output - output_adapter_t oa = nullptr; + OutputSinkType oa; }; } // namespace detail diff --git a/include/nlohmann/detail/output/output_adapters.hpp b/include/nlohmann/detail/output/output_adapters.hpp index 94763e64a..7eb73121c 100644 --- a/include/nlohmann/detail/output/output_adapters.hpp +++ b/include/nlohmann/detail/output/output_adapters.hpp @@ -13,6 +13,7 @@ #include // back_inserter #include // shared_ptr, make_shared #include // basic_string +#include // move #include // vector #ifndef JSON_NO_IO @@ -44,22 +45,32 @@ template struct output_adapter_protocol template using output_adapter_t = std::shared_ptr>; -/// output adapter for byte vectors +/// @brief non-virtual output sink writing into a std::vector +/// +/// This sink is not part of the virtual output_adapter_protocol hierarchy: it is +/// passed to binary_writer by value as a template parameter, so +/// write_character()/write_characters() are ordinary (inlinable) calls with no +/// vtable lookup and no shared_ptr. It is used for the common +/// `to_cbor`/`to_msgpack`/... into a std::vector. output_vector_adapter below +/// wraps this same sink to provide the virtual interface. template> -class output_vector_adapter : public output_adapter_protocol +class output_vector_sink { public: - explicit output_vector_adapter(std::vector& vec) noexcept + explicit output_vector_sink(std::vector& vec) noexcept : v(vec) {} - void write_character(CharType c) override + void write_character(CharType c) { v.push_back(c); } - JSON_HEDLEY_NON_NULL(2) - void write_characters(const CharType* s, std::size_t length) override + // no JSON_HEDLEY_NON_NULL here: binary_writer legitimately passes a null + // pointer with length 0 for empty strings/binary values. Appending an empty + // range is a no-op; the type-erased path tolerates this via the (unattributed) + // virtual base, and the concrete sink must do the same. + void write_characters(const CharType* s, std::size_t length) { v.insert(v.end(), s, s + length); } @@ -68,6 +79,34 @@ class output_vector_adapter : public output_adapter_protocol std::vector& v; }; +/// output adapter for byte vectors +/// +/// The appending itself lives in output_vector_sink; this class only adds the +/// virtual output_adapter_protocol interface on top of it, so both the +/// type-erased and the templated path share one implementation. +template> +class output_vector_adapter : public output_adapter_protocol +{ + public: + explicit output_vector_adapter(std::vector& vec) noexcept + : sink(vec) + {} + + void write_character(CharType c) override + { + sink.write_character(c); + } + + JSON_HEDLEY_NON_NULL(2) + void write_characters(const CharType* s, std::size_t length) override + { + sink.write_characters(s, length); + } + + private: + output_vector_sink sink; +}; + #ifndef JSON_NO_IO /// output adapter for output streams template @@ -118,6 +157,39 @@ class output_string_adapter : public output_adapter_protocol StringType& str; }; +/// @brief output sink forwarding to a type-erased output adapter +/// +/// Wraps the polymorphic output_adapter_t so the same binary_writer template can +/// also target arbitrary adapters (output streams, strings, user-provided +/// adapters) via the `output_adapter`-based overloads. Each write still goes +/// through one virtual call, exactly as before; only the concrete sinks above +/// avoid it. +template +class output_adapter_sink +{ + public: + explicit output_adapter_sink(output_adapter_t adapter) + : oa(std::move(adapter)) + { + JSON_ASSERT(oa); + } + + void write_character(CharType c) + { + oa->write_character(c); + } + + // no JSON_HEDLEY_NON_NULL: forwards (null, 0) for empty payloads, exactly as + // the type-erased path already did before this sink existed + void write_characters(const CharType* s, std::size_t length) + { + oa->write_characters(s, length); + } + + private: + output_adapter_t oa; +}; + template> class output_adapter { diff --git a/include/nlohmann/json.hpp b/include/nlohmann/json.hpp index d55efacc2..a63363fbf 100644 --- a/include/nlohmann/json.hpp +++ b/include/nlohmann/json.hpp @@ -140,7 +140,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec friend ::nlohmann::detail::serializer; template friend class ::nlohmann::detail::iter_impl; - template + template friend class ::nlohmann::detail::binary_writer; template friend class ::nlohmann::detail::binary_reader; @@ -188,6 +188,14 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec template using binary_reader = ::nlohmann::detail::binary_reader; template using binary_writer = ::nlohmann::detail::binary_writer; + // binary_writer over a concrete (non-virtual) sink appending into a std::vector, + // used by the vector-returning to_* overloads + template using vector_binary_writer = + ::nlohmann::detail::binary_writer>; + template static vector_binary_writer vector_writer(std::vector& v) + { + return vector_binary_writer(::nlohmann::detail::output_vector_sink(v)); + } JSON_PRIVATE_UNLESS_TESTED: using serializer = ::nlohmann::detail::serializer; @@ -4444,7 +4452,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec static std::vector to_cbor(const basic_json& j) { std::vector result; - to_cbor(j, result); + result.reserve(detail::binary_reserve_hint(j)); + vector_writer(result).write_cbor(j); return result; } @@ -4467,7 +4476,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec static std::vector to_msgpack(const basic_json& j) { std::vector result; - to_msgpack(j, result); + result.reserve(detail::binary_reserve_hint(j)); + vector_writer(result).write_msgpack(j); return result; } @@ -4492,7 +4502,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool use_type = false) { std::vector result; - to_ubjson(j, result, use_size, use_type); + result.reserve(detail::binary_reserve_hint(j)); + vector_writer(result).write_ubjson(j, use_size, use_type); return result; } @@ -4520,7 +4531,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bjdata_version_t version = bjdata_version_t::draft2) { std::vector result; - to_bjdata(j, result, use_size, use_type, version); + result.reserve(detail::binary_reserve_hint(j)); + vector_writer(result).write_ubjson(j, use_size, use_type, true, true, version); return result; } @@ -4547,7 +4559,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec static std::vector to_bson(const basic_json& j) { std::vector result; - to_bson(j, result); + result.reserve(detail::binary_reserve_hint(j)); + vector_writer(result).write_bson(j); return result; } diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index f300cbdb1..4af4af1e0 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -18621,9 +18621,14 @@ NLOHMANN_JSON_NAMESPACE_END #include // memcpy #include // numeric_limits #include // string +#include // enable_if, is_constructible #include // move #include // vector +#ifdef _MSC_VER + #include // _byteswap_ushort, _byteswap_ulong, _byteswap_uint64 +#endif + // #include // #include @@ -18644,6 +18649,7 @@ NLOHMANN_JSON_NAMESPACE_END #include // back_inserter #include // shared_ptr, make_shared #include // basic_string +#include // move #include // vector #ifndef JSON_NO_IO @@ -18676,22 +18682,32 @@ template struct output_adapter_protocol template using output_adapter_t = std::shared_ptr>; -/// output adapter for byte vectors +/// @brief non-virtual output sink writing into a std::vector +/// +/// This sink is not part of the virtual output_adapter_protocol hierarchy: it is +/// passed to binary_writer by value as a template parameter, so +/// write_character()/write_characters() are ordinary (inlinable) calls with no +/// vtable lookup and no shared_ptr. It is used for the common +/// `to_cbor`/`to_msgpack`/... into a std::vector. output_vector_adapter below +/// wraps this same sink to provide the virtual interface. template> -class output_vector_adapter : public output_adapter_protocol +class output_vector_sink { public: - explicit output_vector_adapter(std::vector& vec) noexcept + explicit output_vector_sink(std::vector& vec) noexcept : v(vec) {} - void write_character(CharType c) override + void write_character(CharType c) { v.push_back(c); } - JSON_HEDLEY_NON_NULL(2) - void write_characters(const CharType* s, std::size_t length) override + // no JSON_HEDLEY_NON_NULL here: binary_writer legitimately passes a null + // pointer with length 0 for empty strings/binary values. Appending an empty + // range is a no-op; the type-erased path tolerates this via the (unattributed) + // virtual base, and the concrete sink must do the same. + void write_characters(const CharType* s, std::size_t length) { v.insert(v.end(), s, s + length); } @@ -18700,6 +18716,34 @@ class output_vector_adapter : public output_adapter_protocol std::vector& v; }; +/// output adapter for byte vectors +/// +/// The appending itself lives in output_vector_sink; this class only adds the +/// virtual output_adapter_protocol interface on top of it, so both the +/// type-erased and the templated path share one implementation. +template> +class output_vector_adapter : public output_adapter_protocol +{ + public: + explicit output_vector_adapter(std::vector& vec) noexcept + : sink(vec) + {} + + void write_character(CharType c) override + { + sink.write_character(c); + } + + JSON_HEDLEY_NON_NULL(2) + void write_characters(const CharType* s, std::size_t length) override + { + sink.write_characters(s, length); + } + + private: + output_vector_sink sink; +}; + #ifndef JSON_NO_IO /// output adapter for output streams template @@ -18750,6 +18794,39 @@ class output_string_adapter : public output_adapter_protocol StringType& str; }; +/// @brief output sink forwarding to a type-erased output adapter +/// +/// Wraps the polymorphic output_adapter_t so the same binary_writer template can +/// also target arbitrary adapters (output streams, strings, user-provided +/// adapters) via the `output_adapter`-based overloads. Each write still goes +/// through one virtual call, exactly as before; only the concrete sinks above +/// avoid it. +template +class output_adapter_sink +{ + public: + explicit output_adapter_sink(output_adapter_t adapter) + : oa(std::move(adapter)) + { + JSON_ASSERT(oa); + } + + void write_character(CharType c) + { + oa->write_character(c); + } + + // no JSON_HEDLEY_NON_NULL: forwards (null, 0) for empty payloads, exactly as + // the type-erased path already did before this sink existed + void write_characters(const CharType* s, std::size_t length) + { + oa->write_characters(s, length); + } + + private: + output_adapter_t oa; +}; + template> class output_adapter { @@ -18796,10 +18873,41 @@ enum class bjdata_version_t // binary writer // /////////////////// +/*! +@brief capacity hint for binary serialization into a std::vector + +Returns a *lower* bound on the number of bytes the serialization will produce, +so that writing an array/object of many elements does not start reallocating +from an empty buffer. Every array element occupies at least one byte in every +supported binary format, and every object entry at least two (a key of at least +one byte plus a value of at least one), plus one byte for the container header, +so the hint can never exceed the final size and the returned vector is never +left holding capacity the caller did not ask for. The buffer still grows +geometrically past the hint, so under-reserving only costs a few later +reallocations. Only the top-level element count is consulted (O(1), no walk of +the DOM); a single scalar, string, or binary value is written in one shot and +needs no hint. +*/ +template +std::size_t binary_reserve_hint(const BasicJsonType& j) +{ + if (j.is_array()) + { + return j.size() + 1; + } + + if (j.is_object()) + { + return (j.size() * 2) + 1; + } + + return 0; +} + /*! @brief serialization to CBOR and MessagePack values */ -template +template> class binary_writer { using string_t = typename BasicJsonType::string_t; @@ -18810,12 +18918,28 @@ class binary_writer /*! @brief create a binary writer + @param[in] sink output sink to write to (a value-type sink such as + output_vector_sink, or output_adapter_sink wrapping a + type-erased output adapter) + */ + explicit binary_writer(OutputSinkType sink) : oa(std::move(sink)) + {} + + /*! + @brief create a binary writer from a type-erased output adapter + + Convenience constructor for the default (output_adapter_sink) sink so the + `output_adapter`-based overloads keep constructing the writer directly from + an adapter. Constrained to sinks that can actually be built from an adapter, + so that a writer over some other sink type is not advertised as constructible + from one. + @param[in] adapter output adapter to write to */ - explicit binary_writer(output_adapter_t adapter) : oa(std::move(adapter)) - { - JSON_ASSERT(oa); - } + template < typename SinkType = OutputSinkType, + typename std::enable_if < std::is_constructible>::value, int >::type = 0 > + explicit binary_writer(output_adapter_t adapter) : oa(SinkType(std::move(adapter))) + {} /*! @param[in] j JSON value to serialize @@ -18856,15 +18980,15 @@ class binary_writer { case value_t::null: { - oa->write_character(to_char_type(0xF6)); + oa.write_character(to_char_type(0xF6)); break; } case value_t::boolean: { - oa->write_character(j.m_data.m_value.boolean - ? to_char_type(0xF5) - : to_char_type(0xF4)); + oa.write_character(j.m_data.m_value.boolean + ? to_char_type(0xF5) + : to_char_type(0xF4)); break; } @@ -18881,22 +19005,22 @@ class binary_writer } else if (j.m_data.m_value.number_integer <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x18)); + oa.write_character(to_char_type(0x18)); write_number(static_cast(j.m_data.m_value.number_integer)); } else if (j.m_data.m_value.number_integer <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x19)); + oa.write_character(to_char_type(0x19)); write_number(static_cast(j.m_data.m_value.number_integer)); } else if (j.m_data.m_value.number_integer <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x1A)); + oa.write_character(to_char_type(0x1A)); write_number(static_cast(j.m_data.m_value.number_integer)); } else { - oa->write_character(to_char_type(0x1B)); + oa.write_character(to_char_type(0x1B)); write_number(static_cast(j.m_data.m_value.number_integer)); } } @@ -18911,22 +19035,22 @@ class binary_writer } else if (positive_number <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x38)); + oa.write_character(to_char_type(0x38)); write_number(static_cast(positive_number)); } else if (positive_number <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x39)); + oa.write_character(to_char_type(0x39)); write_number(static_cast(positive_number)); } else if (positive_number <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x3A)); + oa.write_character(to_char_type(0x3A)); write_number(static_cast(positive_number)); } else { - oa->write_character(to_char_type(0x3B)); + oa.write_character(to_char_type(0x3B)); write_number(static_cast(positive_number)); } } @@ -18941,22 +19065,22 @@ class binary_writer } else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x18)); + oa.write_character(to_char_type(0x18)); write_number(static_cast(j.m_data.m_value.number_unsigned)); } else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x19)); + oa.write_character(to_char_type(0x19)); write_number(static_cast(j.m_data.m_value.number_unsigned)); } else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x1A)); + oa.write_character(to_char_type(0x1A)); write_number(static_cast(j.m_data.m_value.number_unsigned)); } else { - oa->write_character(to_char_type(0x1B)); + oa.write_character(to_char_type(0x1B)); write_number(static_cast(j.m_data.m_value.number_unsigned)); } break; @@ -18967,16 +19091,16 @@ class binary_writer if (std::isnan(j.m_data.m_value.number_float)) { // NaN is 0xf97e00 in CBOR - oa->write_character(to_char_type(0xF9)); - oa->write_character(to_char_type(0x7E)); - oa->write_character(to_char_type(0x00)); + oa.write_character(to_char_type(0xF9)); + oa.write_character(to_char_type(0x7E)); + oa.write_character(to_char_type(0x00)); } else if (std::isinf(j.m_data.m_value.number_float)) { // Infinity is 0xf97c00, -Infinity is 0xf9fc00 - oa->write_character(to_char_type(0xf9)); - oa->write_character(j.m_data.m_value.number_float > 0 ? to_char_type(0x7C) : to_char_type(0xFC)); - oa->write_character(to_char_type(0x00)); + oa.write_character(to_char_type(0xf9)); + oa.write_character(j.m_data.m_value.number_float > 0 ? to_char_type(0x7C) : to_char_type(0xFC)); + oa.write_character(to_char_type(0x00)); } else { @@ -18995,31 +19119,31 @@ class binary_writer } else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x78)); + oa.write_character(to_char_type(0x78)); write_number(static_cast(N)); } else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x79)); + oa.write_character(to_char_type(0x79)); write_number(static_cast(N)); } else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x7A)); + oa.write_character(to_char_type(0x7A)); write_number(static_cast(N)); } // LCOV_EXCL_START else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x7B)); + oa.write_character(to_char_type(0x7B)); write_number(static_cast(N)); } // LCOV_EXCL_STOP // step 2: write the string - oa->write_characters( - reinterpret_cast(j.m_data.m_value.string->data()), - j.m_data.m_value.string->size()); + oa.write_characters( + reinterpret_cast(j.m_data.m_value.string->data()), + j.m_data.m_value.string->size()); break; } @@ -19033,23 +19157,23 @@ class binary_writer } else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x98)); + oa.write_character(to_char_type(0x98)); write_number(static_cast(N)); } else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x99)); + oa.write_character(to_char_type(0x99)); write_number(static_cast(N)); } else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x9A)); + oa.write_character(to_char_type(0x9A)); write_number(static_cast(N)); } // LCOV_EXCL_START else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x9B)); + oa.write_character(to_char_type(0x9B)); write_number(static_cast(N)); } // LCOV_EXCL_STOP @@ -19096,31 +19220,31 @@ class binary_writer } else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x58)); + oa.write_character(to_char_type(0x58)); write_number(static_cast(N)); } else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x59)); + oa.write_character(to_char_type(0x59)); write_number(static_cast(N)); } else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x5A)); + oa.write_character(to_char_type(0x5A)); write_number(static_cast(N)); } // LCOV_EXCL_START else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0x5B)); + oa.write_character(to_char_type(0x5B)); write_number(static_cast(N)); } // LCOV_EXCL_STOP // step 2: write each element - oa->write_characters( - reinterpret_cast(j.m_data.m_value.binary->data()), - N); + oa.write_characters( + reinterpret_cast(j.m_data.m_value.binary->data()), + N); break; } @@ -19135,23 +19259,23 @@ class binary_writer } else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0xB8)); + oa.write_character(to_char_type(0xB8)); write_number(static_cast(N)); } else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0xB9)); + oa.write_character(to_char_type(0xB9)); write_number(static_cast(N)); } else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0xBA)); + oa.write_character(to_char_type(0xBA)); write_number(static_cast(N)); } // LCOV_EXCL_START else if (N <= (std::numeric_limits::max)()) { - oa->write_character(to_char_type(0xBB)); + oa.write_character(to_char_type(0xBB)); write_number(static_cast(N)); } // LCOV_EXCL_STOP @@ -19180,15 +19304,15 @@ class binary_writer { case value_t::null: // nil { - oa->write_character(to_char_type(0xC0)); + oa.write_character(to_char_type(0xC0)); break; } case value_t::boolean: // true and false { - oa->write_character(j.m_data.m_value.boolean - ? to_char_type(0xC3) - : to_char_type(0xC2)); + oa.write_character(j.m_data.m_value.boolean + ? to_char_type(0xC3) + : to_char_type(0xC2)); break; } @@ -19207,25 +19331,25 @@ class binary_writer else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) { // uint 8 - oa->write_character(to_char_type(0xCC)); + oa.write_character(to_char_type(0xCC)); write_number(static_cast(j.m_data.m_value.number_integer)); } else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) { // uint 16 - oa->write_character(to_char_type(0xCD)); + oa.write_character(to_char_type(0xCD)); write_number(static_cast(j.m_data.m_value.number_integer)); } else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) { // uint 32 - oa->write_character(to_char_type(0xCE)); + oa.write_character(to_char_type(0xCE)); write_number(static_cast(j.m_data.m_value.number_integer)); } else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) { // uint 64 - oa->write_character(to_char_type(0xCF)); + oa.write_character(to_char_type(0xCF)); write_number(static_cast(j.m_data.m_value.number_integer)); } } @@ -19240,28 +19364,28 @@ class binary_writer j.m_data.m_value.number_integer <= (std::numeric_limits::max)()) { // int 8 - oa->write_character(to_char_type(0xD0)); + oa.write_character(to_char_type(0xD0)); write_number(static_cast(j.m_data.m_value.number_integer)); } else if (j.m_data.m_value.number_integer >= (std::numeric_limits::min)() && j.m_data.m_value.number_integer <= (std::numeric_limits::max)()) { // int 16 - oa->write_character(to_char_type(0xD1)); + oa.write_character(to_char_type(0xD1)); write_number(static_cast(j.m_data.m_value.number_integer)); } else if (j.m_data.m_value.number_integer >= (std::numeric_limits::min)() && j.m_data.m_value.number_integer <= (std::numeric_limits::max)()) { // int 32 - oa->write_character(to_char_type(0xD2)); + oa.write_character(to_char_type(0xD2)); write_number(static_cast(j.m_data.m_value.number_integer)); } else if (j.m_data.m_value.number_integer >= (std::numeric_limits::min)() && j.m_data.m_value.number_integer <= (std::numeric_limits::max)()) { // int 64 - oa->write_character(to_char_type(0xD3)); + oa.write_character(to_char_type(0xD3)); write_number(static_cast(j.m_data.m_value.number_integer)); } } @@ -19278,25 +19402,25 @@ class binary_writer else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) { // uint 8 - oa->write_character(to_char_type(0xCC)); + oa.write_character(to_char_type(0xCC)); write_number(static_cast(j.m_data.m_value.number_integer)); } else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) { // uint 16 - oa->write_character(to_char_type(0xCD)); + oa.write_character(to_char_type(0xCD)); write_number(static_cast(j.m_data.m_value.number_integer)); } else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) { // uint 32 - oa->write_character(to_char_type(0xCE)); + oa.write_character(to_char_type(0xCE)); write_number(static_cast(j.m_data.m_value.number_integer)); } else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) { // uint 64 - oa->write_character(to_char_type(0xCF)); + oa.write_character(to_char_type(0xCF)); write_number(static_cast(j.m_data.m_value.number_integer)); } break; @@ -19320,26 +19444,26 @@ class binary_writer else if (N <= (std::numeric_limits::max)()) { // str 8 - oa->write_character(to_char_type(0xD9)); + oa.write_character(to_char_type(0xD9)); write_number(static_cast(N)); } else if (N <= (std::numeric_limits::max)()) { // str 16 - oa->write_character(to_char_type(0xDA)); + oa.write_character(to_char_type(0xDA)); write_number(static_cast(N)); } else if (N <= (std::numeric_limits::max)()) { // str 32 - oa->write_character(to_char_type(0xDB)); + oa.write_character(to_char_type(0xDB)); write_number(static_cast(N)); } // step 2: write the string - oa->write_characters( - reinterpret_cast(j.m_data.m_value.string->data()), - j.m_data.m_value.string->size()); + oa.write_characters( + reinterpret_cast(j.m_data.m_value.string->data()), + j.m_data.m_value.string->size()); break; } @@ -19355,13 +19479,13 @@ class binary_writer else if (N <= (std::numeric_limits::max)()) { // array 16 - oa->write_character(to_char_type(0xDC)); + oa.write_character(to_char_type(0xDC)); write_number(static_cast(N)); } else if (N <= (std::numeric_limits::max)()) { // array 32 - oa->write_character(to_char_type(0xDD)); + oa.write_character(to_char_type(0xDD)); write_number(static_cast(N)); } @@ -19417,7 +19541,7 @@ class binary_writer fixed = false; } - oa->write_character(to_char_type(output_type)); + oa.write_character(to_char_type(output_type)); if (!fixed) { write_number(static_cast(N)); @@ -19429,7 +19553,7 @@ class binary_writer ? 0xC8 // ext 16 : 0xC5; // bin 16 - oa->write_character(to_char_type(output_type)); + oa.write_character(to_char_type(output_type)); write_number(static_cast(N)); } else if (N <= (std::numeric_limits::max)()) @@ -19438,7 +19562,7 @@ class binary_writer ? 0xC9 // ext 32 : 0xC6; // bin 32 - oa->write_character(to_char_type(output_type)); + oa.write_character(to_char_type(output_type)); write_number(static_cast(N)); } @@ -19454,9 +19578,9 @@ class binary_writer } // step 2: write the byte string - oa->write_characters( - reinterpret_cast(j.m_data.m_value.binary->data()), - N); + oa.write_characters( + reinterpret_cast(j.m_data.m_value.binary->data()), + N); break; } @@ -19473,13 +19597,13 @@ class binary_writer else if (N <= (std::numeric_limits::max)()) { // map 16 - oa->write_character(to_char_type(0xDE)); + oa.write_character(to_char_type(0xDE)); write_number(static_cast(N)); } else if (N <= (std::numeric_limits::max)()) { // map 32 - oa->write_character(to_char_type(0xDF)); + oa.write_character(to_char_type(0xDF)); write_number(static_cast(N)); } @@ -19518,7 +19642,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('Z')); + oa.write_character(to_char_type('Z')); } break; } @@ -19527,9 +19651,9 @@ class binary_writer { if (add_prefix) { - oa->write_character(j.m_data.m_value.boolean - ? to_char_type('T') - : to_char_type('F')); + oa.write_character(j.m_data.m_value.boolean + ? to_char_type('T') + : to_char_type('F')); } break; } @@ -19556,12 +19680,12 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('S')); + oa.write_character(to_char_type('S')); } write_number_with_ubjson_prefix(j.m_data.m_value.string->size(), true, use_bjdata); - oa->write_characters( - reinterpret_cast(j.m_data.m_value.string->data()), - j.m_data.m_value.string->size()); + oa.write_characters( + reinterpret_cast(j.m_data.m_value.string->data()), + j.m_data.m_value.string->size()); break; } @@ -19569,7 +19693,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('[')); + oa.write_character(to_char_type('[')); } bool prefix_required = true; @@ -19601,14 +19725,14 @@ class binary_writer && !(use_bjdata && std::find(bjdx.begin(), bjdx.end(), first_prefix) != bjdx.end())) { prefix_required = false; - oa->write_character(to_char_type('$')); - oa->write_character(first_prefix); + oa.write_character(to_char_type('$')); + oa.write_character(first_prefix); } } if (use_count) { - oa->write_character(to_char_type('#')); + oa.write_character(to_char_type('#')); write_number_with_ubjson_prefix(j.m_data.m_value.array->size(), true, use_bjdata); } @@ -19619,7 +19743,7 @@ class binary_writer if (!use_count) { - oa->write_character(to_char_type(']')); + oa.write_character(to_char_type(']')); } break; @@ -19629,7 +19753,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('[')); + oa.write_character(to_char_type('[')); } if (use_type && (bjdata_draft3 || !j.m_data.m_value.binary->empty())) @@ -19638,36 +19762,36 @@ class binary_writer { JSON_THROW(other_error::create(502, "use_type requires use_size = true", &j)); } - oa->write_character(to_char_type('$')); - oa->write_character(bjdata_draft3 ? 'B' : 'U'); + oa.write_character(to_char_type('$')); + oa.write_character(bjdata_draft3 ? 'B' : 'U'); } if (use_count) { - oa->write_character(to_char_type('#')); + oa.write_character(to_char_type('#')); write_number_with_ubjson_prefix(j.m_data.m_value.binary->size(), true, use_bjdata); } if (use_type) { - oa->write_characters( - reinterpret_cast(j.m_data.m_value.binary->data()), - j.m_data.m_value.binary->size()); + oa.write_characters( + reinterpret_cast(j.m_data.m_value.binary->data()), + j.m_data.m_value.binary->size()); } else { for (size_t i = 0; i < j.m_data.m_value.binary->size(); ++i) { - oa->write_character(to_char_type(bjdata_draft3 ? 'B' : 'U')); + oa.write_character(to_char_type(bjdata_draft3 ? 'B' : 'U')); // the cast is needed for binary types whose value type // is not an integer (e.g., std::byte) - oa->write_character(to_char_type(static_cast(j.m_data.m_value.binary->data()[i]))); + oa.write_character(to_char_type(static_cast(j.m_data.m_value.binary->data()[i]))); } } if (!use_count) { - oa->write_character(to_char_type(']')); + oa.write_character(to_char_type(']')); } break; @@ -19685,7 +19809,7 @@ class binary_writer if (add_prefix) { - oa->write_character(to_char_type('{')); + oa.write_character(to_char_type('{')); } bool prefix_required = true; @@ -19707,29 +19831,29 @@ class binary_writer if (same_prefix && !(use_bjdata && std::find(bjdx.begin(), bjdx.end(), first_prefix) != bjdx.end())) { prefix_required = false; - oa->write_character(to_char_type('$')); - oa->write_character(first_prefix); + oa.write_character(to_char_type('$')); + oa.write_character(first_prefix); } } if (use_count) { - oa->write_character(to_char_type('#')); + oa.write_character(to_char_type('#')); write_number_with_ubjson_prefix(j.m_data.m_value.object->size(), true, use_bjdata); } for (const auto& el : *j.m_data.m_value.object) { write_number_with_ubjson_prefix(el.first.size(), true, use_bjdata); - oa->write_characters( - reinterpret_cast(el.first.data()), - el.first.size()); + oa.write_characters( + reinterpret_cast(el.first.data()), + el.first.size()); write_ubjson(el.second, use_count, use_type, prefix_required, use_bjdata, bjdata_version); } if (!use_count) { - oa->write_character(to_char_type('}')); + oa.write_character(to_char_type('}')); } break; @@ -19783,13 +19907,13 @@ class binary_writer void write_bson_entry_header(const string_t& name, const std::uint8_t element_type) { - oa->write_character(to_char_type(element_type)); - oa->write_characters( - reinterpret_cast(name.data()), - name.size()); + oa.write_character(to_char_type(element_type)); + oa.write_characters( + reinterpret_cast(name.data()), + name.size()); // the terminating null byte is written explicitly rather than taken // from the buffer, so that string_t::data() need not be null-terminated - oa->write_character(to_char_type(0x00)); + oa.write_character(to_char_type(0x00)); } /*! @@ -19799,7 +19923,7 @@ class binary_writer const bool value) { write_bson_entry_header(name, 0x08); - oa->write_character(value ? to_char_type(0x01) : to_char_type(0x00)); + oa.write_character(value ? to_char_type(0x01) : to_char_type(0x00)); } /*! @@ -19829,12 +19953,12 @@ class binary_writer write_bson_entry_header(name, 0x02); write_number(to_bson_length(value.size() + 1ul), true); - oa->write_characters( - reinterpret_cast(value.data()), - value.size()); + oa.write_characters( + reinterpret_cast(value.data()), + value.size()); // the terminating null byte is written explicitly rather than taken // from the buffer, so that string_t::data() need not be null-terminated - oa->write_character(to_char_type(0x00)); + oa.write_character(to_char_type(0x00)); } /*! @@ -19963,7 +20087,7 @@ class binary_writer write_bson_element(string_t(key.data(), key.size()), el); } - oa->write_character(to_char_type(0x00)); + oa.write_character(to_char_type(0x00)); } /*! @@ -19983,7 +20107,7 @@ class binary_writer write_number(value.has_subtype() ? static_cast(value.subtype()) : static_cast(0x00)); - oa->write_characters(reinterpret_cast(value.data()), value.size()); + oa.write_characters(reinterpret_cast(value.data()), value.size()); } /*! @@ -20109,7 +20233,7 @@ class binary_writer write_bson_element(el.first, el.second); } - oa->write_character(to_char_type(0x00)); + oa.write_character(to_char_type(0x00)); } ////////// @@ -20153,7 +20277,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(get_ubjson_float_prefix(n)); + oa.write_character(get_ubjson_float_prefix(n)); } write_number(n, use_bjdata); } @@ -20169,7 +20293,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('i')); // int8 + oa.write_character(to_char_type('i')); // int8 } write_number(static_cast(n), use_bjdata); } @@ -20177,7 +20301,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('U')); // uint8 + oa.write_character(to_char_type('U')); // uint8 } write_number(static_cast(n), use_bjdata); } @@ -20185,7 +20309,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('I')); // int16 + oa.write_character(to_char_type('I')); // int16 } write_number(static_cast(n), use_bjdata); } @@ -20193,7 +20317,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('u')); // uint16 - bjdata only + oa.write_character(to_char_type('u')); // uint16 - bjdata only } write_number(static_cast(n), use_bjdata); } @@ -20201,7 +20325,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('l')); // int32 + oa.write_character(to_char_type('l')); // int32 } write_number(static_cast(n), use_bjdata); } @@ -20209,7 +20333,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('m')); // uint32 - bjdata only + oa.write_character(to_char_type('m')); // uint32 - bjdata only } write_number(static_cast(n), use_bjdata); } @@ -20217,7 +20341,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('L')); // int64 + oa.write_character(to_char_type('L')); // int64 } write_number(static_cast(n), use_bjdata); } @@ -20225,7 +20349,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('M')); // uint64 - bjdata only + oa.write_character(to_char_type('M')); // uint64 - bjdata only } write_number(static_cast(n), use_bjdata); } @@ -20233,14 +20357,14 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('H')); // high-precision number + oa.write_character(to_char_type('H')); // high-precision number } const auto number = BasicJsonType(n).dump(); write_number_with_ubjson_prefix(number.size(), true, use_bjdata); for (std::size_t i = 0; i < number.size(); ++i) { - oa->write_character(to_char_type(static_cast(number[i]))); + oa.write_character(to_char_type(static_cast(number[i]))); } } } @@ -20257,7 +20381,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('i')); // int8 + oa.write_character(to_char_type('i')); // int8 } write_number(static_cast(n), use_bjdata); } @@ -20265,7 +20389,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('U')); // uint8 + oa.write_character(to_char_type('U')); // uint8 } write_number(static_cast(n), use_bjdata); } @@ -20273,7 +20397,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('I')); // int16 + oa.write_character(to_char_type('I')); // int16 } write_number(static_cast(n), use_bjdata); } @@ -20281,7 +20405,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('u')); // uint16 - bjdata only + oa.write_character(to_char_type('u')); // uint16 - bjdata only } write_number(static_cast(n), use_bjdata); } @@ -20289,7 +20413,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('l')); // int32 + oa.write_character(to_char_type('l')); // int32 } write_number(static_cast(n), use_bjdata); } @@ -20297,7 +20421,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('m')); // uint32 - bjdata only + oa.write_character(to_char_type('m')); // uint32 - bjdata only } write_number(static_cast(n), use_bjdata); } @@ -20305,7 +20429,7 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('L')); // int64 + oa.write_character(to_char_type('L')); // int64 } write_number(static_cast(n), use_bjdata); } @@ -20314,14 +20438,14 @@ class binary_writer { if (add_prefix) { - oa->write_character(to_char_type('H')); // high-precision number + oa.write_character(to_char_type('H')); // high-precision number } const auto number = BasicJsonType(n).dump(); write_number_with_ubjson_prefix(number.size(), true, use_bjdata); for (std::size_t i = 0; i < number.size(); ++i) { - oa->write_character(to_char_type(static_cast(number[i]))); + oa.write_character(to_char_type(static_cast(number[i]))); } } // LCOV_EXCL_STOP @@ -20613,10 +20737,10 @@ class binary_writer } } - oa->write_character('['); - oa->write_character('$'); - oa->write_character(dtype); - oa->write_character('#'); + oa.write_character('['); + oa.write_character('$'); + oa.write_character(dtype); + oa.write_character('#'); key = "_ArraySize_"; write_ubjson(value.at(key), use_count, use_type, true, true, bjdata_version); @@ -20712,6 +20836,87 @@ class binary_writer On the other hand, BSON and BJData use little endian and should reorder on big endian systems. */ + // single-instruction byte swaps (compilers lower these to bswap/rev/movbe); + // used to emit big-endian numbers without a per-byte std::reverse loop + static std::uint16_t byte_swap(std::uint16_t x) noexcept + { +#if defined(__GNUC__) || defined(__clang__) + return __builtin_bswap16(x); +#elif defined(_MSC_VER) + return _byteswap_ushort(x); +#else + return static_cast((x >> 8) | (x << 8)); +#endif + } + + static std::uint32_t byte_swap(std::uint32_t x) noexcept + { +#if defined(__GNUC__) || defined(__clang__) + return __builtin_bswap32(x); +#elif defined(_MSC_VER) + return _byteswap_ulong(x); +#else + return ((x & 0x000000FFu) << 24) | ((x & 0x0000FF00u) << 8) + | ((x & 0x00FF0000u) >> 8) | ((x & 0xFF000000u) >> 24); +#endif + } + + static std::uint64_t byte_swap(std::uint64_t x) noexcept + { +#if defined(__GNUC__) || defined(__clang__) + return __builtin_bswap64(x); +#elif defined(_MSC_VER) + return _byteswap_uint64(x); +#else + x = ((x & 0x00000000FFFFFFFFull) << 32) | ((x & 0xFFFFFFFF00000000ull) >> 32); + x = ((x & 0x0000FFFF0000FFFFull) << 16) | ((x & 0xFFFF0000FFFF0000ull) >> 16); + x = ((x & 0x00FF00FF00FF00FFull) << 8) | ((x & 0xFF00FF00FF00FF00ull) >> 8); + return x; +#endif + } + + /*! + @brief reverse the bytes of a buffer by byte-swapping it as UIntType + + Loading the buffer into an unsigned integer of the same width and swapping + that is what lets the compiler emit a single bswap/rev/movbe; reversing the + buffer element by element does not reliably get there (clang keeps a scalar + shuffle). The two memcpy calls are the only portable way to reinterpret the + bytes and are folded away by every optimizer. + */ + template + static void byte_swap_buffer(std::array& a) noexcept + { + static_assert(sizeof(UIntType) == N, "swap width must match the buffer size"); + UIntType v{}; + std::memcpy(&v, a.data(), sizeof(v)); + v = byte_swap(v); + std::memcpy(a.data(), &v, sizeof(v)); + } + + // reverse the bytes of a fixed-size buffer; a single byte_swap() for the + // common 2/4/8-byte number payloads, std::reverse for any other size + static void reverse_bytes(std::array& a) noexcept + { + byte_swap_buffer(a); + } + + static void reverse_bytes(std::array& a) noexcept + { + byte_swap_buffer(a); + } + + static void reverse_bytes(std::array& a) noexcept + { + byte_swap_buffer(a); + } + + template + static void reverse_bytes(std::array& a) noexcept + { + std::reverse(a.begin(), a.end()); + } + template void write_number(const NumberType n, const bool OutputIsLittleEndian = false) { @@ -20723,10 +20928,10 @@ class binary_writer if (is_little_endian != OutputIsLittleEndian) { // reverse byte order prior to conversion if necessary - std::reverse(vec.begin(), vec.end()); + reverse_bytes(vec); } - oa->write_characters(vec.data(), sizeof(NumberType)); + oa.write_characters(vec.data(), sizeof(NumberType)); } void write_compact_float(const number_float_t n, detail::input_format_t format) @@ -20734,21 +20939,30 @@ class binary_writer #ifdef __GNUC__ JSON_HEDLEY_DIAGNOSTIC_PUSH JSON_HEDLEY_PRAGMA(GCC diagnostic ignored "-Wfloat-equal") +#endif + // When number_float_t is float, static_cast(n) is the identity and + // both branches below are intentionally identical (the "compact" float + // representation is the value itself). Only GCC diagnoses this, and only + // when the sink calls are inlined; clang has no such warning. + // (-Wduplicated-branches only exists from GCC 7 on; naming it on an older + // GCC would itself warn under -Wpragmas) +#if defined(__GNUC__) && !defined(__clang__) && (__GNUC__ >= 7) + JSON_HEDLEY_PRAGMA(GCC diagnostic ignored "-Wduplicated-branches") #endif if (!std::isfinite(n) || ((static_cast(n) >= static_cast(std::numeric_limits::lowest()) && static_cast(n) <= static_cast((std::numeric_limits::max)()) && static_cast(static_cast(n)) == static_cast(n)))) { - oa->write_character(format == detail::input_format_t::cbor - ? get_cbor_float_prefix(static_cast(n)) - : get_msgpack_float_prefix(static_cast(n))); + oa.write_character(format == detail::input_format_t::cbor + ? get_cbor_float_prefix(static_cast(n)) + : get_msgpack_float_prefix(static_cast(n))); write_number(static_cast(n)); } else { - oa->write_character(format == detail::input_format_t::cbor - ? get_cbor_float_prefix(n) - : get_msgpack_float_prefix(n)); + oa.write_character(format == detail::input_format_t::cbor + ? get_cbor_float_prefix(n) + : get_msgpack_float_prefix(n)); write_number(n); } #ifdef __GNUC__ @@ -20815,7 +21029,7 @@ class binary_writer const bool is_little_endian = little_endianness(); /// the output - output_adapter_t oa = nullptr; + OutputSinkType oa; }; } // namespace detail @@ -24156,7 +24370,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec friend ::nlohmann::detail::serializer; template friend class ::nlohmann::detail::iter_impl; - template + template friend class ::nlohmann::detail::binary_writer; template friend class ::nlohmann::detail::binary_reader; @@ -24204,6 +24418,14 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec template using binary_reader = ::nlohmann::detail::binary_reader; template using binary_writer = ::nlohmann::detail::binary_writer; + // binary_writer over a concrete (non-virtual) sink appending into a std::vector, + // used by the vector-returning to_* overloads + template using vector_binary_writer = + ::nlohmann::detail::binary_writer>; + template static vector_binary_writer vector_writer(std::vector& v) + { + return vector_binary_writer(::nlohmann::detail::output_vector_sink(v)); + } JSON_PRIVATE_UNLESS_TESTED: using serializer = ::nlohmann::detail::serializer; @@ -28460,7 +28682,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec static std::vector to_cbor(const basic_json& j) { std::vector result; - to_cbor(j, result); + result.reserve(detail::binary_reserve_hint(j)); + vector_writer(result).write_cbor(j); return result; } @@ -28483,7 +28706,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec static std::vector to_msgpack(const basic_json& j) { std::vector result; - to_msgpack(j, result); + result.reserve(detail::binary_reserve_hint(j)); + vector_writer(result).write_msgpack(j); return result; } @@ -28508,7 +28732,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bool use_type = false) { std::vector result; - to_ubjson(j, result, use_size, use_type); + result.reserve(detail::binary_reserve_hint(j)); + vector_writer(result).write_ubjson(j, use_size, use_type); return result; } @@ -28536,7 +28761,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec const bjdata_version_t version = bjdata_version_t::draft2) { std::vector result; - to_bjdata(j, result, use_size, use_type, version); + result.reserve(detail::binary_reserve_hint(j)); + vector_writer(result).write_ubjson(j, use_size, use_type, true, true, version); return result; } @@ -28563,7 +28789,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec static std::vector to_bson(const basic_json& j) { std::vector result; - to_bson(j, result); + result.reserve(detail::binary_reserve_hint(j)); + vector_writer(result).write_bson(j); return result; } diff --git a/tests/src/unit-binary_writer_sinks.cpp b/tests/src/unit-binary_writer_sinks.cpp new file mode 100644 index 000000000..f60e1bf51 --- /dev/null +++ b/tests/src/unit-binary_writer_sinks.cpp @@ -0,0 +1,198 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#include "doctest_compatibility.h" + +#include +using nlohmann::json; + +#include +#include +#include + +namespace +{ + +// a spread of values exercising every writer path: scalars of each width, the +// float paths, strings, binary, and containers big enough to reallocate +std::vector test_values() +{ + json big_array = json::array(); + for (int i = 0; i < 5000; ++i) + { + big_array.push_back(i); + } + + json big_object = json::object(); + for (int i = 0; i < 1000; ++i) + { + big_object[std::to_string(i)] = i; + } + + return + { + json(nullptr), json(true), json(false), + json(0), json(-1), json(255), json(-129), json(65535), json(-32769), + json(4294967295U), json(-2147483649LL), json(18446744073709551615ULL), + json(0.0), json(-0.5), json(3.1415926535897932), + json(""), json("hello"), json(std::string(1000, 'x')), + json::binary({0x00, 0x01, 0x02}, 42), + json::array(), json::object(), + json::array({1, 2, 3}), json({{"a", 1}, {"b", nullptr}}), + json({{"nested", {{"deep", json::array({1, "two", 3.0, nullptr})}}}}), + big_array, big_object + }; +} + +// values to_bson() accepts: the document must be an object +std::vector bson_values() +{ + json big_object = json::object(); + for (int i = 0; i < 1000; ++i) + { + big_object[std::to_string(i)] = i; + } + + return + { + json::object(), + json({{"a", 1}, {"b", nullptr}, {"c", true}, {"d", 2.5}, {"e", "text"}}), + json({{"arr", json::array({1, 2, 3})}, {"obj", {{"k", "v"}}}}), + big_object + }; +} + +} // namespace + +// The vector-returning to_*(j) overloads write through the non-virtual +// output_vector_sink, while to_*(j, adapter) goes through output_adapter_sink. +// The two are separate code paths that must stay byte-for-byte identical; these +// checks fail if either overload is ever changed without the other. +TEST_CASE("binary writer output sinks") +{ + SECTION("vector sink and adapter sink agree") + { + // note: no SUBCASE inside these loops - doctest keys subcases by + // name/file/line, so a subcase in a loop body would only ever run for + // the first iteration + for (const auto& j : test_values()) + { + CAPTURE(j.dump(-1, ' ', false, json::error_handler_t::replace)); + + std::vector cbor; + json::to_cbor(j, cbor); + CHECK(json::to_cbor(j) == cbor); + + std::vector msgpack; + json::to_msgpack(j, msgpack); + CHECK(json::to_msgpack(j) == msgpack); + + for (const bool use_size : + { + false, true + }) + { + for (const bool use_type : + { + false, true + }) + { + if (use_type && !use_size) + { + continue; // not a supported combination + } + CAPTURE(use_size); + CAPTURE(use_type); + std::vector ubjson; + json::to_ubjson(j, ubjson, use_size, use_type); + CHECK(json::to_ubjson(j, use_size, use_type) == ubjson); + } + } + + for (const auto version : + { + json::bjdata_version_t::draft2, json::bjdata_version_t::draft3 + }) + { + std::vector bjdata; + json::to_bjdata(j, bjdata, false, false, version); + CHECK(json::to_bjdata(j, false, false, version) == bjdata); + } + } + + for (const auto& j : bson_values()) + { + CAPTURE(j.dump()); + std::vector bson; + json::to_bson(j, bson); + CHECK(json::to_bson(j) == bson); + } + } + + SECTION("the char adapter produces the same bytes") + { + for (const auto& j : test_values()) + { + CAPTURE(j.dump(-1, ' ', false, json::error_handler_t::replace)); + + const std::vector expected = json::to_cbor(j); + std::vector as_char; + json::to_cbor(j, as_char); + + REQUIRE(as_char.size() == expected.size()); + std::vector as_bytes; + as_bytes.reserve(as_char.size()); + for (const char c : as_char) + { + as_bytes.push_back(static_cast(c)); + } + CHECK(as_bytes == expected); + } + } +} + +// binary_reserve_hint() is documented as a *lower* bound on the serialized size, +// so that reserving it up front can never leave the returned vector holding +// capacity beyond what the value actually needs. +TEST_CASE("binary_reserve_hint never over-reserves") +{ + for (const auto& j : test_values()) + { + CAPTURE(j.dump(-1, ' ', false, json::error_handler_t::replace)); + + const std::size_t hint = nlohmann::detail::binary_reserve_hint(j); + + CHECK(hint <= json::to_cbor(j).size()); + CHECK(hint <= json::to_msgpack(j).size()); + CHECK(hint <= json::to_ubjson(j).size()); + CHECK(hint <= json::to_ubjson(j, true, true).size()); + CHECK(hint <= json::to_bjdata(j).size()); + } + + for (const auto& j : bson_values()) + { + CAPTURE(j.dump()); + CHECK(nlohmann::detail::binary_reserve_hint(j) <= json::to_bson(j).size()); + } + + SECTION("scalars get no hint") + { + CHECK(nlohmann::detail::binary_reserve_hint(json(nullptr)) == 0); + CHECK(nlohmann::detail::binary_reserve_hint(json(42)) == 0); + CHECK(nlohmann::detail::binary_reserve_hint(json("a string")) == 0); + CHECK(nlohmann::detail::binary_reserve_hint(json::binary({0x01})) == 0); + } + + SECTION("containers are hinted from their element count") + { + CHECK(nlohmann::detail::binary_reserve_hint(json::array()) == 1); + CHECK(nlohmann::detail::binary_reserve_hint(json::array({1, 2, 3})) == 4); + CHECK(nlohmann::detail::binary_reserve_hint(json::object()) == 1); + CHECK(nlohmann::detail::binary_reserve_hint(json({{"a", 1}, {"b", 2}})) == 5); + } +} From 06b0452189ef542ce704a8b4f3b773318d23b361 Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 20:32:31 +0200 Subject: [PATCH 19/64] Bump mkdocs-git-revision-date-localized-plugin in /docs/mkdocs (#5537) Bumps [mkdocs-git-revision-date-localized-plugin](https://github.com/timvink/mkdocs-git-revision-date-localized-plugin) from 1.5.4 to 1.6.0. - [Release notes](https://github.com/timvink/mkdocs-git-revision-date-localized-plugin/releases) - [Commits](https://github.com/timvink/mkdocs-git-revision-date-localized-plugin/compare/v1.5.4...v1.6.0) --- updated-dependencies: - dependency-name: mkdocs-git-revision-date-localized-plugin dependency-version: 1.6.0 dependency-type: direct:production update-type: version-update:semver-minor ... Signed-off-by: dependabot[bot] Co-authored-by: dependabot[bot] <49699333+dependabot[bot]@users.noreply.github.com> --- docs/mkdocs/requirements.txt | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/mkdocs/requirements.txt b/docs/mkdocs/requirements.txt index b1ea71343..6396ec500 100644 --- a/docs/mkdocs/requirements.txt +++ b/docs/mkdocs/requirements.txt @@ -1,7 +1,7 @@ wheel==0.48.0 mkdocs==1.6.1 # documentation framework -mkdocs-git-revision-date-localized-plugin==1.5.4 # plugin "git-revision-date-localized" +mkdocs-git-revision-date-localized-plugin==1.6.0 # plugin "git-revision-date-localized" mkdocs-material==9.7.7 # theme for mkdocs mkdocs-material-extensions==1.3.1 # extensions mkdocs-minify-plugin==0.8.0 # plugin "minify" From f751547a81928c10b5041a6285ab503ad9481b7a Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Wed, 23 Sep 2026 20:43:40 +0200 Subject: [PATCH 20/64] Add missing contributors to the README thanks list (#5543) Add 60 contributors whose work was not yet credited and update seven links that pointed to renamed or reassigned GitHub accounts. Signed-off-by: Niels Lohmann --- README.md | 74 +++++++++++++++++++++++++++++++++++++++++++++++++------ 1 file changed, 67 insertions(+), 7 deletions(-) diff --git a/README.md b/README.md index 4b4236bb6..becce71a1 100644 --- a/README.md +++ b/README.md @@ -1421,7 +1421,7 @@ I deeply appreciate the help of the following people. 6. [Joshua C. Randall](https://github.com/jrandall) fixed a bug in the floating-point serialization. 7. [Aaron Burghardt](https://github.com/aburgh) implemented code to parse streams incrementally. Furthermore, he greatly improved the parser class by allowing the definition of a filter function to discard undesired elements while parsing. 8. [Daniel Kopeček](https://github.com/dkopecek) fixed a bug in the compilation with GCC 5.0. -9. [Florian Weber](https://github.com/Florianjw) fixed a bug in and improved the performance of the comparison operators. +9. [Fiona Johanna Weber](https://github.com/Fiona-J-W) fixed a bug in and improved the performance of the comparison operators. 10. [Eric Cornelius](https://github.com/EricMCornelius) pointed out a bug in the handling with NaN and infinity values. He also improved the performance of the string escaping. 11. [易思龙](https://github.com/likebeta) implemented a conversion from anonymous enums. 12. [kepkin](https://github.com/kepkin) patiently pushed forward the support for Microsoft Visual Studio. @@ -1523,14 +1523,14 @@ I deeply appreciate the help of the following people. 108. [Kevin Tonon](https://github.com/ktonon) overworked the C++11 compiler checks in CMake. 109. [Axel Huebl](https://github.com/ax3l) simplified a CMake check and added support for the [Spack package manager](https://spack.io). 110. [Carlos O'Ryan](https://github.com/coryan) fixed a typo. -111. [James Upjohn](https://github.com/jammehcow) fixed a version number in the compilers section. +111. [James Upjohn](https://github.com/jupjohn) fixed a version number in the compilers section. 112. [Chuck Atkins](https://github.com/chuckatkins) adjusted the CMake files to the CMake packaging guidelines and provided documentation for the CMake integration. 113. [Jan Schöppach](https://github.com/dns13) fixed a typo. 114. [martin-mfg](https://github.com/martin-mfg) fixed a typo. 115. [Matthias Möller](https://github.com/TinyTinni) removed the dependency from `std::stringstream`. 116. [agrianius](https://github.com/agrianius) added code to use alternative string implementations. 117. [Daniel599](https://github.com/Daniel599) allowed to use more algorithms with the `items()` function. -118. [Julius Rakow](https://github.com/jrakow) fixed the Meson include directory and fixed the links to [cppreference.com](https://cppreference.com). +118. [Julius Rakow](https://github.com/juliusrakow) fixed the Meson include directory and fixed the links to [cppreference.com](https://cppreference.com). 119. [Sonu Lohani](https://github.com/sonulohani) fixed the compilation with MSVC 2015 in debug mode. 120. [grembo](https://github.com/grembo) fixed the test suite and re-enabled several test cases. 121. [Hyeon Kim](https://github.com/simnalamburt) introduced the macro `JSON_INTERNAL_CATCH` to control the exception handling inside the library. @@ -1581,7 +1581,7 @@ I deeply appreciate the help of the following people. 166. [Mark Beckwith](https://github.com/wythe) fixed a typo. 167. [yann-morin-1998](https://github.com/yann-morin-1998) helped to reduce the CMake requirement to version 3.1. 168. [Konstantin Podsvirov](https://github.com/podsvirov) maintains a package for the MSYS2 software distro. -169. [remyabel](https://github.com/remyabel) added GNUInstallDirs to the CMake files. +169. [remyabel](https://github.com/remyabel2) added GNUInstallDirs to the CMake files. 170. [Taylor Howard](https://github.com/taylorhoward92) fixed a unit test. 171. [Gabe Ron](https://github.com/Macr0Nerd) implemented the `to_string` method. 172. [Watal M. Iwasaki](https://github.com/heavywatal) fixed a Clang warning. @@ -1608,7 +1608,7 @@ I deeply appreciate the help of the following people. 193. [Hubert Chathi](https://github.com/uhoreg) made CMake's version config file architecture-independent. 194. [OmnipotentEntity](https://github.com/OmnipotentEntity) implemented the binary values for CBOR, MessagePack, BSON, and UBJSON. 195. [ArtemSarmini](https://github.com/ArtemSarmini) fixed a compilation issue with GCC 10 and fixed a leak. -196. [Evgenii Sopov](https://github.com/sea-kg) integrated the library to the wsjcpp package manager. +196. [Evgenii Sopov](https://github.com/sea5kg) integrated the library to the wsjcpp package manager. 197. [Sergey Linev](https://github.com/linev) fixed a compiler warning. 198. [Miguel Magalhães](https://github.com/magamig) fixed the year in the copyright. 199. [Gareth Sylvester-Bradley](https://github.com/garethsb-sony) fixed a compilation issue with MSVC. @@ -1702,7 +1702,7 @@ I deeply appreciate the help of the following people. 287. [NN](https://github.com/NN---) added the Visual Studio output directory to `.gitignore`. 288. [Romain Reignier](https://github.com/romainreignier) improved the performance of the vector output adapter. 289. [Mike](https://github.com/Mike-Leo-Smith) fixed the `std::iterator_traits`. -290. [Richard Hozák](https://github.com/zxey) added macro `JSON_NO_ENUM` to disable default enum conversions. +290. [Richard Hozák](https://github.com/richardhozak) added macro `JSON_NO_ENUM` to disable default enum conversions. 291. [vakokako](https://github.com/vakokako) fixed tests when compiling with C++20. 292. [Alexander “weej” Jones](https://github.com/alexweej) fixed an example in the README. 293. [Eli Schwartz](https://github.com/eli-schwartz) added more files to the `include.zip` archive. @@ -1727,7 +1727,7 @@ I deeply appreciate the help of the following people. 312. [Gareth Sylvester-Bradley](https://github.com/garethsb) added `operator/=` and `operator/` to construct JSON pointers. 313. [Michael Macnair](https://github.com/mykter) added support for afl-fuzz testing. 314. [Berkus Decker](https://github.com/berkus) fixed a typo in the README. -315. [Illia Polishchuk](https://github.com/effolkronium) improved the CMake testing. +315. [Illia Polishchuk](https://github.com/ilqvya) improved the CMake testing. 316. [Ikko Ashimine](https://github.com/eltociear) fixed a typo. 317. [Raphael Grimm](https://github.com/barcode) added the possibility to define a custom base class. 318. [tocic](https://github.com/tocic) fixed typos in the documentation. @@ -1797,6 +1797,66 @@ I deeply appreciate the help of the following people. 382. [bitFiedler](https://github.com/bitFiedler) made GDB pretty printer work with Python 3.8. 383. [Gianfranco Costamagna](https://github.com/LocutusOfBorg) fixed a compiler warning. 384. [risa2000](https://github.com/risa2000) made `std::filesystem::path` conversion to/from UTF-8 encoded string explicit. +385. [AM](https://github.com/maqnouch) fixed typos in the README. +386. [dmenendez-gruposantander](https://github.com/dmenendez-gruposantander) fixed typos in the comments of the examples. +387. [Mihai Stan](https://github.com/mstan-xx) fixed comparisons against the literal `0`. +388. [Matt Gumbel](https://github.com/intelmatt) fixed some `-Weffc++` warnings. +389. [vimpunk](https://github.com/vimpunk) moved a lambda out of an unevaluated context to support older compilers. +390. [Chris Harris](https://github.com/cjh1) fixed the compilation with GCC 4.8. +391. [Palmer Dabbelt](https://github.com/palmer-dabbelt) generated and installed a pkg-config file. +392. [Gus Pozuelo](https://github.com/ap-viavi) made `ordered_map` compatible with GCC 5.5, Clang 3.6, and Xcode 9. +393. [AK](https://github.com/Lioncky) fixed an MSVC build error caused by the `min`/`max` macros from `windows.h`. +394. [Sergiu Deitsch](https://github.com/sergiud) provided a fallback for missing `char8_t` support. +395. [Xiaochuan Ye](https://github.com/XueSongTap) fixed `from_msgpack` for `std::byte` input by specializing `std::char_traits`. +396. [Ville Vesilehto](https://github.com/thevilledev) fixed an overflow in the BJData size calculation and rejected overflowing negative integers in CBOR. +397. [NmPassTHFan](https://github.com/nmpassthf) replaced the deprecated `std::is_trivial` for C++26. +398. [Chris Ever](https://github.com/chirsz-ever) added the `ignore_trailing_commas` parser option. +399. [Kuan-Fu Wu](https://github.com/kfwu1999) fixed the example code for `json_pointer` initialization. +400. [David Kilzer](https://github.com/ddkilzer) added a missing header to the input adapters. +401. [Miko](https://github.com/mikomikotaishi) added proper C++20 module support, simplified the module API, and fixed missing exports. +402. [hitgirl](https://github.com/hitgil) fixed the CMake configuration when cross-compiling. +403. [Devon Thomas](https://github.com/ThomaDevOSU) mentioned the Artistic Style formatting in the contribution guidelines. +404. [Erik Hu](https://github.com/Erikhu1) made Coveralls upload errors non-fatal in the CI. +405. [co63oc](https://github.com/co63oc) fixed typos. +406. [DmitriBogdanov](https://github.com/DmitriBogdanov) fixed broken package manager links in the documentation. +407. [Bander](https://github.com/banderzhm) improved the MSVC compatibility of the C++ modules. +408. [Andy Choi](https://github.com/ccpong) removed an unnecessary `template` keyword before `get` in the README and the documentation. +409. [SamareshSingh](https://github.com/ssam18) fixed single-element brace initialization to copy/move instead of wrapping in an array, fixed the `WITH_DEFAULT` macros for `ordered_map`, and handled moved events in `serve_header.py`. +410. [Aditya](https://github.com/Lumowhisp) improved the documentation of the documentation generation. +411. [cheese1](https://github.com/cheese1) clarified the README. +412. [KhloodElhossiny](https://github.com/khloodelhossiny) enabled `std::string_view` keys in `operator[]`. +413. [Charles Cabergs](https://github.com/cacharle) fixed a `-Wtautological-constant-out-of-range-compare` warning. +414. [EALePain](https://github.com/EALePain) made the `std::tuple` conversion work with reference types such as `std::tie`. +415. [koala_oishi](https://github.com/chibi-dogs) fixed grammatical wording in the README. +416. [riccardoori11](https://github.com/riccardoori11) fixed a typo in the documentation. +417. [Swastik Bose](https://github.com/VasuBhakt) fixed the parent pointers after `update()` with `JSON_DIAGNOSTICS` and fixed the Doxygen autolinking of requirements. +418. [trdesilva](https://github.com/trdesilva) added `front`, `pop_front`, and `push_front` to `json_pointer`. +419. [Akhilesh Arora](https://github.com/akhilesharora) fixed an incomplete-type error with `ordered_json`. +420. [Hariom Phulre](https://github.com/hariomphulre) fixed the C++20 modules compilation with GCC. +421. [Kirill Lokotkov](https://github.com/RUSLoker) fixed printing `long double` values. +422. [George Sedov](https://github.com/radistmorse) added the `NLOHMANN_DEFINE_TYPE_*_WITH_NAMES` macros. +423. [Caillin Nugent](https://github.com/nugentcaillin) added the `NLOHMANN_JSON_SERIALIZE_ENUM_STRICT` macro. +424. [Cosmin D.](https://github.com/drcosmin) fixed `std::filesystem::path` conversions and added an MSVC workaround for `std::unique_ptr`. +425. [Paul Dreik](https://github.com/pauldreik) fixed a test relying on implementation-specific behavior. +426. [Daniel Falk](https://github.com/daniel-falk) added missing copyright notices to the SBOM. +427. [Federico Sfriso](https://github.com/federicosfriso05-dotcom) added support for constructing JSON values from C++20 range views. +428. [Luke Banicevic](https://github.com/banaboi) fixed corrupt BSON output for lengths exceeding `INT32_MAX`, cleaned up the BSON writer, and improved the documentation. +429. [Patrick Armstrong](https://github.com/Patrick10199) updated the CBOR references and the half-precision float assertions. +430. [Yash Bavadiya](https://github.com/xevrion) added checks to all BSON reads. +431. [hum4nBeing](https://github.com/hum4nBeing) fixed the overflow handling of high-precision numbers in UBJSON. +432. [tomatotomata](https://github.com/tomatotomata) added checks for reading CBOR tagged subtypes. +433. [YingqiDuan](https://github.com/YingqiDuan) documented the BSON interoperability. +434. [KBS](https://github.com/youdie006) documented the standards compliance and the strictness of `parse()` and `operator>>`. +435. [Petr Bělohlávek](https://github.com/petrbel) added Clang 21 and 22 to the CI. +436. [Dmitry Rantovov](https://github.com/darkdi) fixed the placement of a CBOR documentation block. +437. [ljcjclljc](https://github.com/ljcjclljc) fixed the comparison of large unsigned integers with signed integers. +438. [Sahil Kamate](https://github.com/sahilkamate03) fixed the handling of CBOR tags 0-5 and 21-23. +439. [Krishnanand G](https://github.com/Krishnanand-G) made the UBJSON writer reject `use_type` without `use_size`. +440. [whn](https://github.com/Whning0513) documented the lenient BSON input handling and corrected the complexity of `to_bson`. +441. [elix3r](https://github.com/22elix3r) fixed `update()` with `merge_objects` when merging a primitive into an object. +442. [Avionic Harshit](https://github.com/avionicharshit-byte) made `diff()` linear when an array shrinks. +443. [Qatadaha Bin Matloob](https://github.com/qatcod) fixed comparisons between integers and floats and fixed unparsable BJData output. +444. [Wu Shuwen](https://github.com/dajiaohuang) removed an unused include. Thanks a lot for helping out! Please [let me know](mailto:mail@nlohmann.me) if I forgot someone. From ed513715a89f288ae571c4a3b84952bf6839b775 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Thu, 24 Sep 2026 06:55:21 +0200 Subject: [PATCH 21/64] Document that a NUL byte in the input is treated as end of input (#5534) * docs: document that a NUL byte in the input is treated as end of input A NUL byte anywhere in the input - trailing, or embedded ahead of more otherwise well-formed JSON - is currently treated the same as genuine end of input, so parsing silently stops there instead of raising the parse_error.101 any other unexpected byte triggers. This mirrors the NUL-terminated-C-string convention already used when no explicit input length is given (json::parse(const char*) already stops at strlen()), just applied uniformly rather than only when a length is genuinely unavailable. This behavior predates this change and is not being altered here - changing it would be an observable, backwards-incompatible behavior change for any caller that (knowingly or not) depends on it, which is not something to do silently in a patch. Documenting the current, verified behavior as a new FAQ entry instead, so it's an intentional and discoverable part of the contract rather than a surprise. Fixes #5530. Signed-off-by: Niels Lohmann Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01N4RQ1Ahan5YAGbnAQGjZTY * Add JSON_STRICT_NUL_HANDLING opt-in macro for issue #5530 A NUL byte anywhere in the input is currently treated the same as real end of input, rather than raising parse_error.101 like any other unexpected byte (documented in the previous commit's FAQ entry). A full unconditional fix was tried in PR #5532 but rejected as too risky to ship by default: any caller could depend on the current behavior, even unknowingly (e.g. a zero-padded buffer). On PR #5534, gregmarr proposed a compile-time opt-in flag instead, and the maintainer agreed, wanting it available now and defaulting to the corrected behavior in 4.0.0. This mirrors the existing JSON_BRACE_INIT_COPY_SEMANTICS precedent as closely as sensible: - JSON_STRICT_NUL_HANDLING defaults to 0 (off); the three lexer sites that treat '\0' as EOF/comment-terminator are gated with `#if !JSON_STRICT_NUL_HANDLING` so the default-off behavior is byte-for-byte identical to today's. - input_adapters.hpp's `T (&array)[N]` overload additionally trims a single trailing '\0' from a `char` array (e.g. a string literal like `json::parse("123")`) when the macro is on, so that case keeps working; every other element type (unsigned char, std::uint8_t, ...) always keeps its full extent. This intentionally does *not* reuse the existing strlen()-based pointer overload via SFINAE-excluding `char` from the array overload, as originally sketched for this change: that approach is ambiguous against the newer generic container overload added since PR #5532, and even where it compiles, strlen()-scanning a `char` array that is not NUL-terminated within its bounds reads past the end of the array (confirmed with AddressSanitizer). Trimming only a single trailing byte, without scanning, avoids both problems. - Documented via docs/mkdocs/docs/api/macros/json_strict_nul_handling.md, linked from the macros index/nav/features page, the FAQ entry, and the parse/accept/operator>> reference pages. - Tested in unit-class_parser.cpp and unit-deserialization.cpp, default state unguarded and opt-in state guarded. Since the library itself #undefs the macro at the end of json.hpp (as JSON_BRACE_INIT_COPY_SEMANTICS already does), a plain `#if defined(JSON_STRICT_NUL_HANDLING)` guard after the include never actually triggers; the tests instead capture the command-line value into a test-local macro before including the header. A few pre-existing fixtures elsewhere (std::array sized one larger than their literal, relying on value-initialization to silently add a trailing zero byte) needed the same one-byte adjustment to keep passing under the opt-in behavior. Unlike the precedent, this adds a proper `JSON_StrictNulHandling` CMake option (rather than a raw -DCMAKE_CXX_FLAGS injection) and wires its ci_test_strict_nul_handling target into the ci_cmake_options job matrix in .github/workflows/ubuntu.yml, so the opt-in build is actually exercised in CI -- closing the one gap in the precedent's own CI setup (ci_test_brace_init_copy_semantics is defined but never referenced by any workflow, so it has never actually run). Co-Authored-By: Claude Sonnet 5 Signed-off-by: Niels Lohmann * Clarify where JSON_STRICT_NUL_HANDLING does not reject NUL bytes Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann Signed-off-by: Niels Lohmann Co-authored-by: Claude Sonnet 5 --- .github/workflows/ubuntu.yml | 2 +- CMakeLists.txt | 6 + cmake/ci.cmake | 17 +++ docs/mkdocs/docs/api/basic_json/accept.md | 8 ++ docs/mkdocs/docs/api/basic_json/parse.md | 8 ++ docs/mkdocs/docs/api/macros/index.md | 5 + .../api/macros/json_strict_nul_handling.md | 126 ++++++++++++++++++ docs/mkdocs/docs/api/operator_gtgt.md | 11 ++ docs/mkdocs/docs/features/macros.md | 13 ++ docs/mkdocs/docs/home/faq.md | 48 +++++++ docs/mkdocs/docs/integration/cmake.md | 5 + docs/mkdocs/mkdocs.yml | 1 + .../nlohmann/detail/input/input_adapters.hpp | 15 +++ include/nlohmann/detail/input/lexer.hpp | 13 +- include/nlohmann/detail/macro_scope.hpp | 4 + include/nlohmann/detail/macro_unscope.hpp | 1 + single_include/nlohmann/json.hpp | 33 ++++- tests/src/unit-class_parser.cpp | 108 ++++++++++++++- tests/src/unit-deserialization.cpp | 56 +++++++- tests/src/unit-regression2.cpp | 6 +- 20 files changed, 475 insertions(+), 11 deletions(-) create mode 100644 docs/mkdocs/docs/api/macros/json_strict_nul_handling.md diff --git a/.github/workflows/ubuntu.yml b/.github/workflows/ubuntu.yml index 7ac4cfcb0..c7804079c 100644 --- a/.github/workflows/ubuntu.yml +++ b/.github/workflows/ubuntu.yml @@ -100,7 +100,7 @@ jobs: container: ubuntu:focal strategy: matrix: - target: [ci_cmake_flags, ci_test_diagnostics, ci_test_diagnostic_positions, ci_test_noexceptions, ci_test_noimplicitconversions, ci_test_legacycomparison, ci_test_noglobaludls, ci_test_disableenumserialization, ci_test_skiplibraryversioncheck, ci_test_simdutf] + target: [ci_cmake_flags, ci_test_diagnostics, ci_test_diagnostic_positions, ci_test_noexceptions, ci_test_noimplicitconversions, ci_test_legacycomparison, ci_test_noglobaludls, ci_test_disableenumserialization, ci_test_skiplibraryversioncheck, ci_test_simdutf, ci_test_strict_nul_handling] steps: - name: Install build-essential run: apt-get update ; apt-get install -y build-essential unzip wget git libssl-dev diff --git a/CMakeLists.txt b/CMakeLists.txt index 9669946a5..4c43c23ff 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -59,6 +59,7 @@ option(JSON_LegacyDiscardedValueComparison "Enable legacy discarded value compar option(JSON_Install "Install CMake targets during install step." ${MAIN_PROJECT}) option(JSON_MultipleHeaders "Use non-amalgamated version of the library." ON) option(JSON_SystemInclude "Include as system headers (skip for clang-tidy)." OFF) +option(JSON_StrictNulHandling "Build with strict NUL-byte handling enabled." OFF) if (JSON_CI) include(ci) @@ -108,6 +109,10 @@ if (JSON_Diagnostics) message(STATUS "Diagnostics enabled (JSON_DIAGNOSTICS=1)") endif() +if (JSON_StrictNulHandling) + message(STATUS "Strict NUL-byte handling enabled (JSON_STRICT_NUL_HANDLING=1)") +endif() + if (JSON_Diagnostic_Positions) message(STATUS "Diagnostic positions enabled (JSON_DIAGNOSTIC_POSITIONS=1)") endif() @@ -141,6 +146,7 @@ target_compile_definitions( $<$:JSON_DIAGNOSTICS=1> $<$:JSON_DIAGNOSTIC_POSITIONS=1> $<$:JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON=1> + $<$:JSON_STRICT_NUL_HANDLING=1> ) target_include_directories( diff --git a/cmake/ci.cmake b/cmake/ci.cmake index 7d085fb2b..2d4bbf6d2 100644 --- a/cmake/ci.cmake +++ b/cmake/ci.cmake @@ -245,6 +245,23 @@ add_custom_target(ci_test_brace_init_copy_semantics COMMENT "Compile and test with brace-init copy semantics enabled" ) +############################################################################### +# Enable strict NUL-byte handling. +############################################################################### + +add_custom_target(ci_test_strict_nul_handling + COMMAND ${CMAKE_COMMAND} + -DCMAKE_BUILD_TYPE=Debug -GNinja + -DJSON_BuildTests=ON -DJSON_FastTests=ON -DJSON_StrictNulHandling=ON + -S${PROJECT_SOURCE_DIR} -B${PROJECT_BINARY_DIR}/build_strict_nul_handling + COMMAND ${CMAKE_COMMAND} --build ${PROJECT_BINARY_DIR}/build_strict_nul_handling + # unit-testsuites contains a fixture (a "1e308" test value) that relies on the + # legacy NUL-as-end-of-input behavior this macro disables; exclude it here, as + # it is expected to fail under strict NUL handling and is out of scope for it + COMMAND cd ${PROJECT_BINARY_DIR}/build_strict_nul_handling && ${CMAKE_CTEST_COMMAND} --parallel ${N} --output-on-failure -E "test-testsuites" + COMMENT "Compile and test with strict NUL-byte handling enabled" +) + ############################################################################### # Disable global UDLs. ############################################################################### diff --git a/docs/mkdocs/docs/api/basic_json/accept.md b/docs/mkdocs/docs/api/basic_json/accept.md index 5ba009c3e..0cdcae3a8 100644 --- a/docs/mkdocs/docs/api/basic_json/accept.md +++ b/docs/mkdocs/docs/api/basic_json/accept.md @@ -90,6 +90,10 @@ Linear in the length of the input. The parser is a predictive LL(1) parser. A UTF-8 byte order mark is silently ignored. +By default, a `'\0'` (NUL) byte anywhere in the input is treated as end of input, rather than as an ordinary (and, +outside of a string, invalid) byte; see the [FAQ entry](../../home/faq.md#nul-bytes-in-the-input) for details and the +[`JSON_STRICT_NUL_HANDLING`](../macros/json_strict_nul_handling.md) macro to opt into rejecting it instead. + ## Examples ??? example @@ -111,6 +115,8 @@ A UTF-8 byte order mark is silently ignored. - [parse](parse.md) - deserialize from a compatible input - [sax_parse](sax_parse.md) - parse input using the SAX interface - [operator>>](../operator_gtgt.md) - deserialize from stream +- [`JSON_STRICT_NUL_HANDLING`](../macros/json_strict_nul_handling.md) - opt in to rejecting a NUL byte in the input + instead of treating it as end of input ## Version history @@ -120,6 +126,8 @@ A UTF-8 byte order mark is silently ignored. - Added `ignore_trailing_commas` in version 3.13.0. - Extended container support (1) to include types with lvalue-only ADL `begin`/`end` (matching `std::begin`/`std::end` semantics) in version 3.13.0. - Extended overload (2) to accept heterogeneous iterator+sentinel pairs (C++20 ranges support) in version 3.13.0. +- `JSON_STRICT_NUL_HANDLING` added in version 3.13.0 to optionally reject a NUL byte in the input instead of treating + it as end of input; planned to become the default in version 4.0.0. !!! warning "Deprecation" diff --git a/docs/mkdocs/docs/api/basic_json/parse.md b/docs/mkdocs/docs/api/basic_json/parse.md index ef22e9873..20bb1c708 100644 --- a/docs/mkdocs/docs/api/basic_json/parse.md +++ b/docs/mkdocs/docs/api/basic_json/parse.md @@ -103,6 +103,10 @@ A UTF-8 byte order mark is silently ignored. Invalid Unicode escapes and unpaired surrogates in the input are reported as [`parse_error.101`](../../home/exceptions.md#jsonexceptionparse_error101) with a detailed message. +By default, a `'\0'` (NUL) byte anywhere in the input is treated as end of input, rather than as an ordinary (and, +outside of a string, invalid) byte; see the [FAQ entry](../../home/faq.md#nul-bytes-in-the-input) for details and the +[`JSON_STRICT_NUL_HANDLING`](../macros/json_strict_nul_handling.md) macro to opt into rejecting it instead. + ## Examples ??? example "Parsing from a character array" @@ -236,6 +240,8 @@ Invalid Unicode escapes and unpaired surrogates in the input are reported as - [accept](accept.md) - check if the input is valid JSON - [sax_parse](sax_parse.md) - parse input using the SAX interface - [operator>>](../operator_gtgt.md) - deserialize from stream +- [`JSON_STRICT_NUL_HANDLING`](../macros/json_strict_nul_handling.md) - opt in to rejecting a NUL byte in the input + instead of treating it as end of input ## Version history @@ -246,6 +252,8 @@ Invalid Unicode escapes and unpaired surrogates in the input are reported as - Added `ignore_trailing_commas` in version 3.13.0. - Extended container support (1) to include types with lvalue-only ADL `begin`/`end` (matching `std::begin`/`std::end` semantics) in version 3.13.0. - Extended overload (2) to accept heterogeneous iterator+sentinel pairs (C++20 ranges support) in version 3.13.0. +- `JSON_STRICT_NUL_HANDLING` added in version 3.13.0 to optionally reject a NUL byte in the input instead of treating + it as end of input; planned to become the default in version 4.0.0. !!! warning "Deprecation" diff --git a/docs/mkdocs/docs/api/macros/index.md b/docs/mkdocs/docs/api/macros/index.md index e818f032a..ca3af720c 100644 --- a/docs/mkdocs/docs/api/macros/index.md +++ b/docs/mkdocs/docs/api/macros/index.md @@ -14,6 +14,11 @@ header. See also the [macro overview page](../../features/macros.md). - [**JSON_DIAGNOSTIC_POSITIONS**](json_diagnostic_positions.md) - access positions of elements - [**JSON_NOEXCEPTION**](json_noexception.md) - switch off exceptions +## Parsing + +- [**JSON_STRICT_NUL_HANDLING**](json_strict_nul_handling.md) - opt in to rejecting a NUL byte in the input instead of + treating it as end of input + ## Language support - [**JSON_HAS_CPP_11**
**JSON_HAS_CPP_14**
**JSON_HAS_CPP_17**
**JSON_HAS_CPP_20**](json_has_cpp_11.md) - set supported C++ standard diff --git a/docs/mkdocs/docs/api/macros/json_strict_nul_handling.md b/docs/mkdocs/docs/api/macros/json_strict_nul_handling.md new file mode 100644 index 000000000..1832f2353 --- /dev/null +++ b/docs/mkdocs/docs/api/macros/json_strict_nul_handling.md @@ -0,0 +1,126 @@ +# JSON_STRICT_NUL_HANDLING + +```cpp +#define JSON_STRICT_NUL_HANDLING /* value */ +``` + +When defined to `1`, a `'\0'` (NUL) byte in JSON text input is rejected with `parse_error.101`, like any other +unexpected byte, instead of being silently treated as end of input. + +The macro only affects the JSON text parser ([`parse`](../basic_json/parse.md), [`accept`](../basic_json/accept.md), +[`sax_parse`](../basic_json/sax_parse.md), and [`operator>>`](../operator_gtgt.md)). There are three cases where a NUL +byte is still not rejected: + +- The binary formats ([`from_bjdata`](../basic_json/from_bjdata.md), [`from_bson`](../basic_json/from_bson.md), + [`from_cbor`](../basic_json/from_cbor.md), [`from_msgpack`](../basic_json/from_msgpack.md), + [`from_ubjson`](../basic_json/from_ubjson.md)) are never affected: there, `0x00` is ordinary data. +- A bare `const char*` pointer has no length of its own, so its length is still determined with `strlen()`. The first + NUL byte therefore still marks the end of the input, and nothing after it is read. +- One trailing `'\0'` at the end of a `char` array (e.g., a string literal) is trimmed; see the warning below. + +## Default definition + +The default value is `0` (disabled — existing behavior is preserved). + +```cpp +#define JSON_STRICT_NUL_HANDLING 0 +``` + +## Notes + +!!! note "Background" + + By default, a `'\0'` byte anywhere in the input is treated the same as the real end of the input, rather than as + an ordinary (and, outside of a string, invalid) byte. Everything from that byte onward is silently ignored, + without a parse error - including further, otherwise well-formed JSON: + + ```cpp + json::parse(std::string("123") + '\0'); // == 123, no error + json::parse(std::string("123") + '\0' + "true"); // == 123, the "true" is silently ignored too + ``` + + This falls out of the same convention used when no explicit input length is given at all: parsing from a + `const char*` already stops at the first NUL byte via `strlen()`, since a bare pointer has no length of its own. + The library applies that same NUL-terminated-C-string convention uniformly, rather than only when a length is + genuinely unavailable - so a `std::string`, iterator range, or container whose content happens to include a NUL + byte is affected the same way a raw `const char*` would be (see the + [FAQ entry](../../home/faq.md#nul-bytes-in-the-input) for a fuller explanation). + + This was not fixed unconditionally, because doing so is backwards-incompatible for any caller who happens to + depend on the current behavior - even unknowingly, for instance because their input already contains trailing + padding they never noticed was being discarded (see [#5530](https://github.com/nlohmann/json/issues/5530)). + This macro instead offers an opt-in path to the corrected behavior ahead of version 4.0.0, where it is planned to + become the default. + +!!! warning "Opt-in only" + + This macro must be defined **before** including ``. Defining it after the include has no + effect. + + Enabling it also changes how a `char` array (including a string literal, e.g. `json::parse("123")`) is read: such + an array normally carries a trailing `'\0'` contributed by the compiler, not by the source text. With this macro + enabled, that one trailing byte is trimmed if present so that parsing a string literal keeps working; every other + byte in the array - including any `'\0'` that is not the very last element - is read as real data and rejected + like any other unexpected byte. Arrays of any other element type (`unsigned char`, `std::uint8_t`, ...), as used + for CBOR or MessagePack, are never affected by this trimming; their full extent - including a genuine trailing + `0x00` - is always preserved, in both states of this macro. + +!!! tip "Workaround without the macro" + + To reject a NUL byte without enabling this macro, trim your input yourself before calling `parse()`: + + ```cpp + s.resize(s.find('\0')); // drop everything from the first NUL onward, if any + json::parse(s); + ``` + +## Examples + +??? example "Default behavior (macro not defined)" + + Without the macro, a NUL byte silently ends parsing at that point: + + ```cpp + #include + + using json = nlohmann::json; + + int main() + { + json j = json::parse(std::string("123") + '\0' + "true"); + // j is 123 -- the '\0' and everything after it is silently ignored + } + ``` + +??? example "Opt-in strict handling (macro defined to 1)" + + With the macro, a NUL byte is rejected like any other unexpected byte: + + ```cpp + #define JSON_STRICT_NUL_HANDLING 1 + #include + + using json = nlohmann::json; + + int main() + { + json j = json::parse(std::string("123") + '\0' + "true"); + // throws parse_error.101 -- the NUL byte is now invalid input, + // exactly like any other unexpected trailing byte + + json ok = json::parse("123"); + // ok is 123 -- parsing from a string literal still works + } + ``` + +## See also + +- [FAQ: NUL bytes in the input](../../home/faq.md#nul-bytes-in-the-input) +- [**parse**](../basic_json/parse.md) - deserialize from a compatible input +- [**accept**](../basic_json/accept.md) - check if the input is valid JSON +- [**operator>>**](../operator_gtgt.md) - deserialize from stream + +## Version history + +- Added in version 3.13.0. +- Planned to become the default (with the macro removed) in version 4.0.0. diff --git a/docs/mkdocs/docs/api/operator_gtgt.md b/docs/mkdocs/docs/api/operator_gtgt.md index a9ba17564..3e60d5236 100644 --- a/docs/mkdocs/docs/api/operator_gtgt.md +++ b/docs/mkdocs/docs/api/operator_gtgt.md @@ -72,6 +72,13 @@ input >> j2; // parses the next value Note that reading concatenated values does **not** work for [JSON Lines](../features/parsing/json_lines.md) (newline-delimited JSON) input -- see that page for why and for the recommended alternative. +By default, a `'\0'` (NUL) byte encountered while reading a value is treated as end of input, rather than as an +ordinary (and, outside of a string, invalid) byte; see the [FAQ entry](../home/faq.md#nul-bytes-in-the-input) for +details and the [`JSON_STRICT_NUL_HANDLING`](macros/json_strict_nul_handling.md) macro to opt into rejecting it +instead. Because `operator>>` only parses a single value and does not require the rest of the stream to be consumed, +a NUL byte *after* a complete value has no effect on `operator>>` either way; it only matters while a value is still +being read. + !!! warning "Deprecation" This function replaces function `#!cpp std::istream& operator<<(basic_json& j, std::istream& i)` which has @@ -98,7 +105,11 @@ Note that reading concatenated values does **not** work for [JSON Lines](../feat - [accept](basic_json/accept.md) - check if the input is valid JSON - [parse](basic_json/parse.md) - deserialize from a compatible input +- [`JSON_STRICT_NUL_HANDLING`](macros/json_strict_nul_handling.md) - opt in to rejecting a NUL byte in the input + instead of treating it as end of input ## Version history - Added in version 1.0.0. +- `JSON_STRICT_NUL_HANDLING` added in version 3.13.0 to optionally reject a NUL byte in the input instead of treating + it as end of input; planned to become the default in version 4.0.0. diff --git a/docs/mkdocs/docs/features/macros.md b/docs/mkdocs/docs/features/macros.md index c4602fa5a..927e00df3 100644 --- a/docs/mkdocs/docs/features/macros.md +++ b/docs/mkdocs/docs/features/macros.md @@ -105,6 +105,19 @@ using the library with compilers that do not fully support C++11 and may only wo See [full documentation of `JSON_SKIP_UNSUPPORTED_COMPILER_CHECK`](../api/macros/json_skip_unsupported_compiler_check.md). +## `JSON_STRICT_NUL_HANDLING` + +When defined to `1`, a `'\0'` (NUL) byte anywhere in the input is rejected with `parse_error.101`, like any other +unexpected byte, instead of being silently treated as end of input (see the +[FAQ entry](../home/faq.md#nul-bytes-in-the-input) for background). The default value is `0`, which preserves the +existing behavior; this is planned to become the default in version 4.0.0. + +The strict handling can also be enabled with the CMake option +[`JSON_StrictNulHandling`](../integration/cmake.md#json_strictnulhandling) (`OFF` by default) which sets +`JSON_STRICT_NUL_HANDLING` accordingly. + +See [full documentation of `JSON_STRICT_NUL_HANDLING`](../api/macros/json_strict_nul_handling.md). + ## `JSON_THROW_USER(exception)` This macro overrides `#!cpp throw` calls inside the library. The argument is the exception to be thrown. diff --git a/docs/mkdocs/docs/home/faq.md b/docs/mkdocs/docs/home/faq.md index 8394dcfc7..8b3602bd1 100644 --- a/docs/mkdocs/docs/home/faq.md +++ b/docs/mkdocs/docs/home/faq.md @@ -90,6 +90,54 @@ The library supports **Unicode input** as follows: In most cases, the parser is right to complain, because the input is not UTF-8 encoded. This is especially true for Microsoft Windows, where Latin-1 or ISO 8859-1 is often the standard encoding. +### NUL bytes in the input + +!!! question "Questions" + + - Why does `json::parse()` silently ignore part of my input? + - Why does a `std::string`/buffer with extra data after the JSON text parse without error, while a similar-looking string with extra text does not? + +A `'\0'` (NUL) byte anywhere in the input is treated the same as the real end of the input, rather than as an ordinary (and, outside of a string, invalid) byte. Everything from that byte onward is silently ignored, without a parse error — including further, otherwise well-formed JSON: + +```cpp +json::parse(std::string("123") + '\0'); // == 123, no error +json::parse(std::string("123") + '\0' + "true"); // == 123, the "true" is silently ignored too +``` + +This is different from any other unexpected trailing byte, which *does* raise [`parse_error.101`](../home/exceptions.md#jsonexceptionparse_error101): + +```cpp +json::parse("123x"); // throws parse_error.101: unexpected additional data +``` + +This falls out of the same convention used when no explicit input length is given at all: `json::parse(const char*)` already stops at the first NUL byte via `strlen()`, since a bare pointer has no length of its own. The library applies that same NUL-terminated-C-string convention uniformly, rather than only when a length is genuinely unavailable — so a `std::string`, iterator range, or container whose content happens to include a NUL byte is affected the same way a raw `const char*` would be. + +If your input may contain a trailing or embedded NUL that is **not** meant to signal the end of the JSON text — for instance, a fixed-size, zero-padded buffer — trim it yourself before calling `parse()`, since the library will otherwise silently stop there instead of raising an error: + +```cpp +s.resize(s.find('\0')); // drop everything from the first NUL onward, if any +json::parse(s); +``` + +**Opt-in strict handling (since version 3.13.0)** + +Manually trimming every input is easy to forget. If you define [`JSON_STRICT_NUL_HANDLING`](../api/macros/json_strict_nul_handling.md) to `1` before including the library, a `'\0'` byte is instead rejected like any other unexpected byte and raises `parse_error.101`, instead of being treated as end of input: + +```cpp +#define JSON_STRICT_NUL_HANDLING 1 +#include + +json::parse(std::string("123") + '\0'); // throws parse_error.101 instead of silently returning 123 +``` + +This macro defaults to `0` (disabled, preserving the behavior described above) to avoid breaking existing code that may depend on it, even unknowingly; it is planned to become the default in version 4.0.0. See [its documentation](../api/macros/json_strict_nul_handling.md) for details, including how it also affects `char` arrays such as string literals. + +Note that this is unrelated to an *unescaped* NUL byte occurring **inside** a quoted JSON string, which is a different, already-invalid case and is correctly rejected either way: + +```cpp +json::parse(std::string("\"") + '\0' + "\""); // throws parse_error.101: control character U+0000 (NUL) must be escaped to \u0000 +``` + ### Wide string handling !!! question diff --git a/docs/mkdocs/docs/integration/cmake.md b/docs/mkdocs/docs/integration/cmake.md index a8a6d52b6..71512cbd5 100644 --- a/docs/mkdocs/docs/integration/cmake.md +++ b/docs/mkdocs/docs/integration/cmake.md @@ -198,6 +198,11 @@ Use the non-amalgamated version of the library. This option is `ON` by default. Treat the library headers like system headers (i.e., adding `SYSTEM` to the [`target_include_directories`](https://cmake.org/cmake/help/latest/command/target_include_directories.html) call) to check for this library by tools like Clang-Tidy. This option is `OFF` by default. +### `JSON_StrictNulHandling` + +Reject a `'\0'` (NUL) byte in the input instead of treating it as end of input, by defining the macro +[`JSON_STRICT_NUL_HANDLING`](../api/macros/json_strict_nul_handling.md). This option is `OFF` by default. + ### `JSON_Valgrind` Execute the test suite with [Valgrind](https://valgrind.org). This option is `OFF` by default. Depends on `JSON_BuildTests`. diff --git a/docs/mkdocs/mkdocs.yml b/docs/mkdocs/mkdocs.yml index d0f9cfdfd..856d86e6d 100644 --- a/docs/mkdocs/mkdocs.yml +++ b/docs/mkdocs/mkdocs.yml @@ -294,6 +294,7 @@ nav: - 'JSON_NO_IO': api/macros/json_no_io.md - 'JSON_SKIP_LIBRARY_VERSION_CHECK': api/macros/json_skip_library_version_check.md - 'JSON_SKIP_UNSUPPORTED_COMPILER_CHECK': api/macros/json_skip_unsupported_compiler_check.md + - 'JSON_STRICT_NUL_HANDLING': api/macros/json_strict_nul_handling.md - 'JSON_USE_GLOBAL_UDLS': api/macros/json_use_global_udls.md - 'JSON_USE_IMPLICIT_CONVERSIONS': api/macros/json_use_implicit_conversions.md - 'JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON': api/macros/json_use_legacy_discarded_value_comparison.md diff --git a/include/nlohmann/detail/input/input_adapters.hpp b/include/nlohmann/detail/input/input_adapters.hpp index bd19d32a8..775a8398c 100644 --- a/include/nlohmann/detail/input/input_adapters.hpp +++ b/include/nlohmann/detail/input/input_adapters.hpp @@ -762,6 +762,21 @@ contiguous_bytes_input_adapter input_adapter(CharT b) template auto input_adapter(T (&array)[N]) -> decltype(input_adapter(array, array + N)) // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) { +#if JSON_STRICT_NUL_HANDLING + // A `char` array from string-literal initialization (e.g. json::parse("123")) + // carries a trailing '\0' contributed by the compiler, not by the source + // text; drop exactly that one byte so it is not mistaken for real trailing + // data. Every other element type (unsigned char, std::uint8_t, ...) keeps + // the full extent unconditionally, since a trailing zero byte there is + // genuine data (e.g. CBOR/MessagePack). This intentionally does not + // strlen()-scan the array (as the pointer overload above does for a + // null-delimited string): for a `char` array that is not NUL-terminated + // within its bounds, that would read past the end of the array. + if (std::is_same::type, char>::value && N > 0 && array[N - 1] == 0) + { + return input_adapter(array, array + N - 1); + } +#endif return input_adapter(array, array + N); } diff --git a/include/nlohmann/detail/input/lexer.hpp b/include/nlohmann/detail/input/lexer.hpp index bc31337f9..fe85cd53d 100644 --- a/include/nlohmann/detail/input/lexer.hpp +++ b/include/nlohmann/detail/input/lexer.hpp @@ -952,7 +952,9 @@ class lexer : public lexer_base case '\n': case '\r': case char_traits::eof(): +#if !JSON_STRICT_NUL_HANDLING case '\0': +#endif return true; default: @@ -970,8 +972,10 @@ class lexer : public lexer_base { switch (get()) { - case char_traits::eof(): +#if !JSON_STRICT_NUL_HANDLING case '\0': +#endif + case char_traits::eof(): { error_message = "invalid comment; missing closing '*/'"; return false; @@ -2153,9 +2157,12 @@ scan_number_done: case '9': return scan_number_dispatch(std::integral_constant {}); - // end of input (the null byte is needed when parsing from - // string literals) +#if !JSON_STRICT_NUL_HANDLING case '\0': +#endif + // end of input; by default, a null byte is also treated as end of + // input for backwards compatibility (see JSON_STRICT_NUL_HANDLING + // to opt into rejecting a null byte in the input instead) case char_traits::eof(): return token_type::end_of_input; diff --git a/include/nlohmann/detail/macro_scope.hpp b/include/nlohmann/detail/macro_scope.hpp index 4682fd361..8aacc0c51 100644 --- a/include/nlohmann/detail/macro_scope.hpp +++ b/include/nlohmann/detail/macro_scope.hpp @@ -807,3 +807,7 @@ void templated_json_throw(ExceptionType exception) #ifndef JSON_BRACE_INIT_COPY_SEMANTICS #define JSON_BRACE_INIT_COPY_SEMANTICS 0 #endif + +#ifndef JSON_STRICT_NUL_HANDLING + #define JSON_STRICT_NUL_HANDLING 0 +#endif diff --git a/include/nlohmann/detail/macro_unscope.hpp b/include/nlohmann/detail/macro_unscope.hpp index 2ac25a46d..c692ea68e 100644 --- a/include/nlohmann/detail/macro_unscope.hpp +++ b/include/nlohmann/detail/macro_unscope.hpp @@ -27,6 +27,7 @@ #undef JSON_DISABLE_ENUM_SERIALIZATION #undef JSON_USE_GLOBAL_UDLS #undef JSON_BRACE_INIT_COPY_SEMANTICS +#undef JSON_STRICT_NUL_HANDLING #ifndef JSON_TEST_KEEP_MACROS #undef JSON_CATCH diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 4af4af1e0..4418a6c19 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -3186,6 +3186,10 @@ void templated_json_throw(ExceptionType exception) #define JSON_BRACE_INIT_COPY_SEMANTICS 0 #endif +#ifndef JSON_STRICT_NUL_HANDLING + #define JSON_STRICT_NUL_HANDLING 0 +#endif + #if JSON_HAS_THREE_WAY_COMPARISON #include // partial_ordering #endif @@ -7903,6 +7907,21 @@ contiguous_bytes_input_adapter input_adapter(CharT b) template auto input_adapter(T (&array)[N]) -> decltype(input_adapter(array, array + N)) // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) { +#if JSON_STRICT_NUL_HANDLING + // A `char` array from string-literal initialization (e.g. json::parse("123")) + // carries a trailing '\0' contributed by the compiler, not by the source + // text; drop exactly that one byte so it is not mistaken for real trailing + // data. Every other element type (unsigned char, std::uint8_t, ...) keeps + // the full extent unconditionally, since a trailing zero byte there is + // genuine data (e.g. CBOR/MessagePack). This intentionally does not + // strlen()-scan the array (as the pointer overload above does for a + // null-delimited string): for a `char` array that is not NUL-terminated + // within its bounds, that would read past the end of the array. + if (std::is_same::type, char>::value && N > 0 && array[N - 1] == 0) + { + return input_adapter(array, array + N - 1); + } +#endif return input_adapter(array, array + N); } @@ -9512,7 +9531,9 @@ class lexer : public lexer_base case '\n': case '\r': case char_traits::eof(): +#if !JSON_STRICT_NUL_HANDLING case '\0': +#endif return true; default: @@ -9530,8 +9551,10 @@ class lexer : public lexer_base { switch (get()) { - case char_traits::eof(): +#if !JSON_STRICT_NUL_HANDLING case '\0': +#endif + case char_traits::eof(): { error_message = "invalid comment; missing closing '*/'"; return false; @@ -10713,9 +10736,12 @@ scan_number_done: case '9': return scan_number_dispatch(std::integral_constant {}); - // end of input (the null byte is needed when parsing from - // string literals) +#if !JSON_STRICT_NUL_HANDLING case '\0': +#endif + // end of input; by default, a null byte is also treated as end of + // input for backwards compatibility (see JSON_STRICT_NUL_HANDLING + // to opt into rejecting a null byte in the input instead) case char_traits::eof(): return token_type::end_of_input; @@ -30027,6 +30053,7 @@ struct formatter // NOLINT(cert-dcl58-c #undef JSON_DISABLE_ENUM_SERIALIZATION #undef JSON_USE_GLOBAL_UDLS #undef JSON_BRACE_INIT_COPY_SEMANTICS +#undef JSON_STRICT_NUL_HANDLING #ifndef JSON_TEST_KEEP_MACROS #undef JSON_CATCH diff --git a/tests/src/unit-class_parser.cpp b/tests/src/unit-class_parser.cpp index df4e7270d..e532992f4 100644 --- a/tests/src/unit-class_parser.cpp +++ b/tests/src/unit-class_parser.cpp @@ -8,6 +8,14 @@ #include "doctest_compatibility.h" +// capture whether JSON_STRICT_NUL_HANDLING was enabled on the command line +// (e.g. -DJSON_STRICT_NUL_HANDLING=1) *before* including json.hpp, since the +// library #undefs JSON_STRICT_NUL_HANDLING itself once the header has been +// fully processed (see include/nlohmann/detail/macro_unscope.hpp) +#if defined(JSON_STRICT_NUL_HANDLING) && (JSON_STRICT_NUL_HANDLING == 1) + #define JSON_TEST_STRICT_NUL_HANDLING_ENABLED 1 +#endif + #define JSON_TESTS_PRIVATE #include using nlohmann::json; @@ -545,6 +553,88 @@ TEST_CASE("parser class") } } + SECTION("NUL byte handling (issue #5530, JSON_STRICT_NUL_HANDLING)") + { + // by default, a NUL byte anywhere in the input (not inside a quoted + // string, which is covered above) is silently treated the same as + // real end of input; JSON_STRICT_NUL_HANDLING (off by default, see + // docs/mkdocs/docs/api/macros/json_strict_nul_handling.md) makes a + // NUL byte an error like any other unexpected byte instead. + // + // The two sections below are mutually exclusive: this whole test + // binary is compiled once, with JSON_STRICT_NUL_HANDLING either + // left at its default or forced to 1 (e.g. by the dedicated + // ci_test_strict_nul_handling CI target), so only the section + // matching the actual, compiled-in behavior can pass. +#if !defined(JSON_TEST_STRICT_NUL_HANDLING_ENABLED) + SECTION("default behavior (macro not enabled)") + { + // a NUL byte after a complete value silently truncates the input + std::string s = "123"; + s.push_back('\0'); + s += "4"; + CHECK(json::parse(s) == json(123)); + CHECK(json::accept(s)); + + // parsing from a string literal is unaffected either way + CHECK(json::parse("123") == json(123)); + } +#endif + +#if defined(JSON_TEST_STRICT_NUL_HANDLING_ENABLED) + SECTION("opt-in strict behavior (JSON_STRICT_NUL_HANDLING == 1)") + { + // a NUL byte after a complete value is now a parse error, + // instead of silently truncating the input + { + std::string s = "123"; + s.push_back('\0'); + json _; // NOLINT(readability-identifier-naming) + CHECK_THROWS_WITH_AS(_ = json::parse(s), + "[json.exception.parse_error.101] parse error at line 1, column 4: syntax error while parsing value - invalid literal; last read: '123'; expected end of input", + json::parse_error&); + CHECK_FALSE(json::accept(s)); + } + + // a NUL byte where a value is expected is now a parse error, + // instead of being treated the same as an empty input + { + const std::string s(1, '\0'); + json _; // NOLINT(readability-identifier-naming) + CHECK_THROWS_WITH_AS(_ = json::parse(s), + "[json.exception.parse_error.101] parse error at line 1, column 1: syntax error while parsing value - invalid literal; last read: ''", + json::parse_error&); + CHECK_FALSE(json::accept(s)); + } + + // a NUL byte inside a // comment no longer stops the comment + // scan early; scanning continues correctly past it + { + std::string s = "1 // a"; + s.push_back('\0'); + s += "b\n"; + CHECK(json::parse(s, nullptr, true, true) == json(1)); + CHECK(json::accept(s, true, true)); + } + + // a NUL byte inside a /* */ comment no longer stops the + // comment scan early either + { + std::string s = "1 /* a"; + s.push_back('\0'); + s += "b */ "; + CHECK(json::parse(s, nullptr, true, true) == json(1)); + CHECK(json::accept(s, true, true)); + } + + // regression guard: parsing from a string literal (which + // carries a compiler-appended trailing '\0') still works, + // even though a NUL byte is now rejected everywhere else + CHECK(json::parse("123") == json(123)); + } +#endif + } + SECTION("number") { SECTION("integers") @@ -1894,7 +1984,13 @@ TEST_CASE("parser class") SECTION("from std::array") { - std::array v { {'t', 'r', 'u', 'e'} }; + // NOTE: this array is sized to exactly the length of "true" (unlike + // the trailing-NUL-tolerant default behavior elsewhere in this file, + // see the "NUL byte handling" section above); a size of 5 here would + // leave a value-initialized trailing 0x00 element that is only + // silently accepted as end-of-input by default and would fail under + // JSON_STRICT_NUL_HANDLING + std::array v { {'t', 'r', 'u', 'e'} }; json j; json::parser(nlohmann::detail::input_adapter(std::begin(v), std::end(v))).parse(true, j); CHECK(j == json(true)); @@ -2035,7 +2131,17 @@ TEST_CASE("parser class") { json _; CHECK_THROWS_WITH_AS(_ = json::parse("/a", nullptr, true, true), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid comment; expecting '/' or '*' after '/'; last read: '/a'", json::parse_error); + // "/*" is a string literal, so it carries a compiler-appended trailing + // '\0'; by default that NUL is read like any other byte and shows up + // in "last read", but JSON_STRICT_NUL_HANDLING trims exactly that one + // trailing byte from a char array (see + // docs/mkdocs/docs/api/macros/json_strict_nul_handling.md), so it no + // longer appears in the message in that state +#if defined(JSON_TEST_STRICT_NUL_HANDLING_ENABLED) + CHECK_THROWS_WITH_AS(_ = json::parse("/*", nullptr, true, true), "[json.exception.parse_error.101] parse error at line 1, column 3: syntax error while parsing value - invalid comment; missing closing '*/'; last read: '/*'", json::parse_error); +#else CHECK_THROWS_WITH_AS(_ = json::parse("/*", nullptr, true, true), "[json.exception.parse_error.101] parse error at line 1, column 3: syntax error while parsing value - invalid comment; missing closing '*/'; last read: '/*'", json::parse_error); +#endif } #if JSON_DIAGNOSTIC_POSITIONS diff --git a/tests/src/unit-deserialization.cpp b/tests/src/unit-deserialization.cpp index 77d49c08b..0dbfdd15c 100644 --- a/tests/src/unit-deserialization.cpp +++ b/tests/src/unit-deserialization.cpp @@ -8,6 +8,14 @@ #include "doctest_compatibility.h" +// capture whether JSON_STRICT_NUL_HANDLING was enabled on the command line +// (e.g. -DJSON_STRICT_NUL_HANDLING=1) *before* including json.hpp, since the +// library #undefs JSON_STRICT_NUL_HANDLING itself once the header has been +// fully processed (see include/nlohmann/detail/macro_unscope.hpp) +#if defined(JSON_STRICT_NUL_HANDLING) && (JSON_STRICT_NUL_HANDLING == 1) + #define JSON_TEST_STRICT_NUL_HANDLING_ENABLED 1 +#endif + #include using nlohmann::json; #ifdef JSON_TEST_NO_GLOBAL_UDLS @@ -323,6 +331,23 @@ TEST_CASE("deserialization") CHECK(j == json({"foo", 1, 2, 3, false, {{"one", 1}}})); } + SECTION("operator>> with a NUL byte after the value (issue #5530)") + { + // operator>> parses non-strictly (it does not require the whole + // stream to be consumed), so a NUL byte following a complete + // value is simply left unread on the stream and never reaches + // the "expected end of input" check that JSON_STRICT_NUL_HANDLING + // affects; this holds regardless of the macro (verified below for + // the opt-in state as well) + std::string data = "123"; + data.push_back('\0'); + std::istringstream ss(data); + json j; + ss >> j; + CHECK(j == json(123)); + CHECK(ss.good()); + } + SECTION("user-defined string literal") { CHECK("[\"foo\",1,2,3,false,{\"one\":1}]"_json == json({"foo", 1, 2, 3, false, {{"one", 1}}})); @@ -405,6 +430,27 @@ TEST_CASE("deserialization") CHECK_THROWS_WITH_AS(ss >> j, "[json.exception.parse_error.101] parse error at line 1, column 29: syntax error while parsing array - unexpected end of input; expected ']'", json::parse_error&); } +#if defined(JSON_TEST_STRICT_NUL_HANDLING_ENABLED) + SECTION("operator>> with a NUL byte where a value is expected (JSON_STRICT_NUL_HANDLING == 1, issue #5530)") + { + // a trailing NUL byte *after* a complete value is unaffected by the + // macro (see the successful-deserialization "operator>> with a NUL + // byte after the value" section above): operator>> parses + // non-strictly and never reaches the "expected end of input" check + // that the macro changes. A NUL byte where a *value* is expected, + // however, goes through the same token dispatch as any other input + // and is affected: with the macro enabled it now raises + // parse_error.101 (like any other unrecognized byte) instead of + // being silently treated the same as an empty stream. + std::string const data(1, '\0'); + std::istringstream ss(data); + json j; + CHECK_THROWS_WITH_AS(ss >> j, + "[json.exception.parse_error.101] parse error at line 1, column 1: syntax error while parsing value - invalid literal; last read: ''", + json::parse_error&); + } +#endif + SECTION("user-defined string literal") { CHECK_THROWS_WITH_AS("[\"foo\",1,2,3,false,{\"one\":1}"_json, "[json.exception.parse_error.101] parse error at line 1, column 29: syntax error while parsing array - unexpected end of input; expected ']'", json::parse_error&); @@ -453,7 +499,11 @@ TEST_CASE("deserialization") SECTION("from std::array") { - std::array const v { {'t', 'r', 'u', 'e'} }; + // sized to exactly the length of "true": a size of 5 would leave + // a value-initialized trailing 0x00 element that is only + // silently accepted as end-of-input by default and would fail + // under JSON_STRICT_NUL_HANDLING + std::array const v { {'t', 'r', 'u', 'e'} }; CHECK(json::parse(v) == json(true)); CHECK(json::accept(v)); @@ -549,7 +599,9 @@ TEST_CASE("deserialization") SECTION("from std::array") { - std::array v { {'t', 'r', 'u', 'e'} }; + // sized to exactly the length of "true", see the analogous + // "from std::array" section above for why + std::array v { {'t', 'r', 'u', 'e'} }; CHECK(json::parse(std::begin(v), std::end(v)) == json(true)); CHECK(json::accept(std::begin(v), std::end(v))); diff --git a/tests/src/unit-regression2.cpp b/tests/src/unit-regression2.cpp index 2c0cf6549..b128b7a73 100644 --- a/tests/src/unit-regression2.cpp +++ b/tests/src/unit-regression2.cpp @@ -606,7 +606,11 @@ TEST_CASE("regression tests 2") SECTION("issue #2546 - parsing containers of std::byte") { const char DATA[] = R"("Hello, world!")"; // NOLINT(misc-const-correctness,cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) - const auto s = std::as_bytes(std::span(DATA)); + // exclude the trailing '\0' that string-literal initialization adds to + // DATA: std::span(DATA) would span the full array extent (including + // that NUL), which is only silently accepted as end-of-input by default + // and would fail under JSON_STRICT_NUL_HANDLING + const auto s = std::as_bytes(std::span(DATA, sizeof(DATA) - 1)); const json j = json::parse(s); CHECK(j.dump() == "\"Hello, world!\""); } From 56b3ee566c63e087c827e7b5accec2eb6bb74cff Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Thu, 24 Sep 2026 17:01:15 +0200 Subject: [PATCH 22/64] Fix stack overflow when copying a deeply nested value (#5387) (#5389) * Bound the descent of the copy constructor basic_json's copy constructor copied objects and arrays by handing the container to its own copy constructor, which copy-constructs every element and so reaches this constructor again, once per nesting level. A value nested deeply enough exhausted the call stack and terminated the process with a segmentation fault - no exception, nothing the caller could catch. Parsing such a value works, as the parser is iterative, and so does destroying one, as #1436 made destruction iterative. Bound how far the copy descends rather than take the call stack away from it. The first levels are copied exactly as they were - the containers copy their own elements, which is by far the fastest way to fill them - and only once the copy has descended 128 levels is the value below it finished without the call stack, through an explicit worklist. Copying can therefore no longer exhaust the stack, however deeply a value is nested, while a value nested less deeply than the bound - all but a vanishing minority - is copied by the very same code as before and pays only for one counter. That counter lives in thread_local storage, as one shared between threads would be raced. JSON_NO_THREAD_LOCAL switches it off for toolchains without thread_local; copying then goes through the worklist right away, which yields the same values but is measurably slower. The deferred values are completed before the copy they belong to returns, so a value copied while another copy is going on - by a custom base class, say - is unaffected by the copy it is nested in. operator= takes its argument by value, so copy assignment is fixed as well. Copying is as fast as it was, within measurement noise (medians of 9 interleaved runs, clang -O3): -1.3% for an array of strings, +0.0% for a flat object, +0.1% for a flat array of numbers, +0.3% for nested arrays, +0.6% for nested objects and +1.2% for a twitter-like document. Copying a three-key object costs about ten nanoseconds more, the counter. Deferring every level instead, rather than only those below the bound, measured between 3% and 9% slower depending on the shape of the value. This fixes #5387 for the copy constructor. dump() is still recursive. Signed-off-by: Niels Lohmann * Test the copy constructor's iterative path in CI The copy constructor descends into 128 levels before it finishes a value without the call stack, so the iterative path is otherwise only reached by the few tests that nest deeper than that. JSON_NO_THREAD_LOCAL switches the descent off, which sends every value down that path. Running the whole test suite that way covers it with every object type, string type, allocator, and base class the suite already exercises. The new ci_test_no_thread_local target does that; the macro had no build coverage at all before. Copying a nested value also has to carry over what the element-wise copy constructor would have copied: the parents that JSON_DIAGNOSTICS relies on, and the positions that JSON_DIAGNOSTIC_POSITIONS reports. Both are now checked on either side of the descent bound, for objects and arrays. Neither was tested before, and dropping either one makes the new tests fail. Also quantify what JSON_NO_THREAD_LOCAL costs a copy instead of calling it "measurably slower". Signed-off-by: Niels Lohmann * Split the regression tests so that they keep linking Linking test-regression2 fails with "relocation truncated to fit: IMAGE_REL_AMD64_REL32 against `.rdata'" once its object grows past what the MinGW linker copes with, and the copy constructor's helpers push it over: the object grows by 6.3%, from 4,654,128 to 4,944,920 bytes at -O0, and develop links at the smaller of the two. Building the tests optimized shrinks the object enough to link, but the binaries clang 11.0.1 and clang 18.1.8 then produce crash before doctest prints its first line - 39 of 102 tests on clang 18 - so the objects have to become smaller rather than denser. Moving the test cases that follow "regression tests 2" into a file of their own brings that object to 4,687,888 bytes, which is 0.7% above the size that links today rather than 6.3%. Both files still build for C++11, C++17 and C++20, and run the same 9 test cases and 135 assertions as before, now spread over two binaries. New regression tests belong in unit-regression3.cpp from here on, which is what CONTRIBUTING.md now says. Signed-off-by: Niels Lohmann * Do not use thread_local storage with Clang targeting MinGW Every test that copies a value segfaults there - 42 of 105 on clang 11.0.1, 39 of 102 on clang 18.1.8 - while the same tests pass with GCC targeting MinGW, with Clang targeting MSVC, and with every other toolchain the library is tested on. The counter that bounds the copy constructor's descent is the library's first use of thread_local, so that job had never exercised it before. JSON_NO_THREAD_LOCAL already covers toolchains without thread_local storage, and copying yields the same values with it, only more slowly. Define it for this one automatically. Signed-off-by: Niels Lohmann * Balance the warning suppression the split separated unit-regression2.cpp opens a DOCTEST_CLANG_SUPPRESS_WARNING_PUSH block at the top and closed it at the very bottom, which the split moved into unit-regression3.cpp: one file was left with a push and no pop, the other with a pop and no push, which clang reports as an error. Give each file the pair it needs. Signed-off-by: Niels Lohmann * Check both shapes without a C-style array clang-tidy rejects the array the two shapes were iterated over (cppcoreguidelines-avoid-c-arrays). The array only existed because astyle reformats a range-for over a braced initializer list into something unreadable; naming the two cases avoids both. Signed-off-by: Niels Lohmann * Split the regression tests far enough to leave room The first split left unit-regression2.cpp 0.7% below the size develop links at, which the comparison change in the follow-up immediately used up: the MinGW linker fails on test-regression2_cpp20 again, naming copy_shallow and to_partial_ordering among the relocations it cannot fit. Move the sections from "issue #2067" on, and the helper types they use, so that the file stops being the one that decides whether the tests can be linked at all. At -O0 and C++20, unit-regression2.cpp is now 2,964,944 bytes against develop's 4,708,248, and 3,070,568 bytes with the follow-up applied - roughly a third smaller either way, rather than a fraction of a percent larger. The 135 assertions are the same ones as before, now spread over three test cases in two files. Also silence the clang-tidy findings the deep-nesting tests draw: the copies they make are what is being tested, and the reserve() computation gets its parentheses. Signed-off-by: Niels Lohmann * Move the #4804 alias to the file that uses it The split left the json_4804 alias behind in unit-regression2.cpp while the test case that uses it went to unit-regression3.cpp, which does not build for C++17 and C++20 as a result. Signed-off-by: Niels Lohmann * Include where the split moved its only use The #2546 test case guards itself with __has_include(), but the include itself sat in unit-regression2.cpp's preamble and stayed behind, so the section compiled without a declaration wherever the guard passed - which nvhpc reported and libc++ builds do not, as they skip the section altogether. Signed-off-by: Niels Lohmann * Keep the descent bookkeeping in one place Copying carried a depth count, a depth limit and a guard of its own, and the comparison in the follow-up added a second set beside them. Neither operation needs its own: they are never nested inside one another by the library - copying a value does not compare one, and comparing two values does not copy them - and where user code nests them anyway, sharing the count only ends a descent sooner than it had to. So there is now one nesting_depth(), one nesting_depth_limit() and one nesting_depth_guard, which the follow-up uses instead of adding its own. Inverting the test in copy_structured leaves the too-deep case and the no-thread-local case as the same code. The guard takes the count rather than looking it up, because the caller has looked it up already to test it against the limit, and reaching thread-local storage twice on the path that is taken almost every time is worth avoiding. The switch that copies the value of anything that is not an object or an array was written twice - once in the copy constructor, once in copy_shallow - so that adding a value_t meant editing both, and missing one would have been silent. It is copy_leaf_value now, and inlined: both callers have already sorted the containers out, and folding that test into the switch is what keeps a value made mostly of numbers copying as fast as it did. Copying canada.json, citm_catalog.json and twitter.json is within 0.6% of what it was before, measured as a paired ratio over 18 interleaved rounds against a run-to-run spread of 0.3%. Signed-off-by: Niels Lohmann * Check that an abandoned copy can still be destroyed Copying a value without the call stack builds the copy from the top down, and every value whose own copy has not been made yet stays a null value until it is. That is what lets a copy be abandoned half-built: the destructor finds nothing but complete values and null ones. Nothing tested it. Failing an allocation part-way through a copy of a deeply nested value does, with the allocator the file already has for exactly this kind of test. Signed-off-by: Niels Lohmann * Name the test's locals so Flawfinder stops matching them The code scanning job reports CWE-362 - "check when opening files" - for a test that opens no files: Flawfinder matched a local variable called open. Rename it and its partner. Signed-off-by: Niels Lohmann * Keep the descent guard's bookkeeping self-contained nesting_depth_limit() and nesting_depth_guard were only used inside the JSON_NO_THREAD_LOCAL-guarded branch of copy_structured(), but were defined unconditionally. Move them inside the #ifndef, and have the guard look up the depth and test it against the limit itself (via okay()) instead of making the caller do it - the caller no longer needs to touch nesting_depth() at all. Also shrink the thread-local counter to std::uint8_t, matching what its own doc comment already argued. Addresses gregmarr's review comments on #5389. Signed-off-by: Niels Lohmann * Make nesting_depth_guard usable regardless of JSON_NO_THREAD_LOCAL nesting_depth_limit() and nesting_depth() stay behind #ifndef JSON_NO_THREAD_LOCAL, since a descent cannot be bounded without a per-thread count. But the guard itself now always exists, becoming a no-op that is never okay() under that macro - the same way the bound is already reached on every call without one. copy_structured() no longer needs to know which case it is in. This is what lets #5390 reuse the guard for comparison, which cannot test JSON_NO_THREAD_LOCAL where the macro-based operators use it: the guard now carries that distinction itself instead of requiring every caller to. Signed-off-by: Niels Lohmann * Silence VS2015's C4503 for the custom-base-class test The deep-copy support added for #5387 lengthened the mangled name of std::allocator_traits<...>::construct for the test's map type past VS2015's limit, which /WX turns into a build failure even though the name is only used for (now-truncated) debug info. Signed-off-by: Niels Lohmann * Remove dead unused-parameter casts from copy_metadata() @gregmarr asked whether the static_cast pair in the JSON_DIAGNOSTIC_POSITIONS-off branch was needed for an empty json_base_class_t. It isn't: src and dst are already referenced unconditionally by the base-class copy above, so no -Wunused-parameter warning fires either way (checked with -Wall -Wextra -Wunused-parameter, JSON_DIAGNOSTIC_POSITIONS 0 and 1). Signed-off-by: Niels Lohmann * Fix CI: build custom array types without a fill constructor, re-amalgamate copy_array_level() built the destination array with the fill constructor array_t(count, value), which is not part of the array container interface the library otherwise assumes (e.g. custom ArrayTypes that only provide a default and an iterator-pair constructor, as covered by unit-custom-array-type.cpp). Default- construct the array and resize() it instead, matching how the rest of the codebase already grows array_t. Also re-run the amalgamation, which had fallen out of sync with include/nlohmann/json.hpp. Signed-off-by: Niels Lohmann Co-Authored-By: Claude Sonnet 5 --------- Signed-off-by: Niels Lohmann Co-authored-by: Claude Sonnet 5 --- .github/workflows/ubuntu.yml | 2 +- .github/workflows/windows.yml | 4 + cmake/ci.cmake | 19 + docs/mkdocs/docs/api/macros/index.md | 1 + .../docs/api/macros/json_no_thread_local.md | 47 ++ docs/mkdocs/docs/features/macros.md | 7 + docs/mkdocs/mkdocs.yml | 1 + include/nlohmann/detail/macro_scope.hpp | 9 + include/nlohmann/json.hpp | 411 ++++++++++++++--- single_include/nlohmann/json.hpp | 420 +++++++++++++++--- tests/CMakeLists.txt | 7 +- tests/src/unit-allocator.cpp | 51 +++ tests/src/unit-diagnostic-positions.cpp | 66 +++ tests/src/unit-diagnostics.cpp | 57 +++ tests/src/unit-large_json.cpp | 151 +++++++ tests/src/unit-ordered_json.cpp | 34 ++ 16 files changed, 1175 insertions(+), 112 deletions(-) create mode 100644 docs/mkdocs/docs/api/macros/json_no_thread_local.md diff --git a/.github/workflows/ubuntu.yml b/.github/workflows/ubuntu.yml index c7804079c..d8c27c621 100644 --- a/.github/workflows/ubuntu.yml +++ b/.github/workflows/ubuntu.yml @@ -100,7 +100,7 @@ jobs: container: ubuntu:focal strategy: matrix: - target: [ci_cmake_flags, ci_test_diagnostics, ci_test_diagnostic_positions, ci_test_noexceptions, ci_test_noimplicitconversions, ci_test_legacycomparison, ci_test_noglobaludls, ci_test_disableenumserialization, ci_test_skiplibraryversioncheck, ci_test_simdutf, ci_test_strict_nul_handling] + target: [ci_cmake_flags, ci_test_diagnostics, ci_test_diagnostic_positions, ci_test_noexceptions, ci_test_noimplicitconversions, ci_test_legacycomparison, ci_test_noglobaludls, ci_test_disableenumserialization, ci_test_skiplibraryversioncheck, ci_test_simdutf, ci_test_strict_nul_handling, ci_test_no_thread_local] steps: - name: Install build-essential run: apt-get update ; apt-get install -y build-essential unzip wget git libssl-dev diff --git a/.github/workflows/windows.yml b/.github/workflows/windows.yml index 068c5a0f1..99a8aa3ca 100644 --- a/.github/workflows/windows.yml +++ b/.github/workflows/windows.yml @@ -158,6 +158,10 @@ jobs: # to fit: IMAGE_REL_AMD64_SECREL against `.debug_line'" because the # MinGW linker cannot relocate the debug sections this test produces. # The tests are only built and run here, so the debug info is not used. + # Do not add -O1 here to shrink the objects further: it does make them + # link, but the binaries clang 11.0.1 and clang 18.1.8 then produce crash + # before doctest prints its first line - 39 of 102 tests on clang 18. + # Keep the objects small by splitting the test files instead. - name: Run CMake run: cmake -S . -B build ^ -DCMAKE_CXX_COMPILER="C:/Program Files/LLVM/bin/clang++.exe" ^ diff --git a/cmake/ci.cmake b/cmake/ci.cmake index 2d4bbf6d2..f854138b6 100644 --- a/cmake/ci.cmake +++ b/cmake/ci.cmake @@ -311,6 +311,25 @@ add_custom_target(ci_test_skiplibraryversioncheck COMMENT "Compile and run a translation unit simulating a mismatched library version, with JSON_SKIP_LIBRARY_VERSION_CHECK defined" ) +############################################################################### +# Disable thread-local storage. +############################################################################### + +# Without thread-local storage, the copy constructor cannot bound its descent +# and copies every object and array without the call stack. That path is +# otherwise only reached by values nested deeper than the bound, so this target +# is what runs the whole test suite through it. +add_custom_target(ci_test_no_thread_local + COMMAND ${CMAKE_COMMAND} + -DCMAKE_BUILD_TYPE=Debug -GNinja + -DJSON_BuildTests=ON + -DCMAKE_CXX_FLAGS=-DJSON_NO_THREAD_LOCAL + -S${PROJECT_SOURCE_DIR} -B${PROJECT_BINARY_DIR}/build_no_thread_local + COMMAND ${CMAKE_COMMAND} --build ${PROJECT_BINARY_DIR}/build_no_thread_local + COMMAND cd ${PROJECT_BINARY_DIR}/build_no_thread_local && ${CMAKE_CTEST_COMMAND} --parallel ${N} --output-on-failure + COMMENT "Compile and test without thread-local storage" +) + ############################################################################### # Coverage. ############################################################################### diff --git a/docs/mkdocs/docs/api/macros/index.md b/docs/mkdocs/docs/api/macros/index.md index ca3af720c..70f02a7a6 100644 --- a/docs/mkdocs/docs/api/macros/index.md +++ b/docs/mkdocs/docs/api/macros/index.md @@ -27,6 +27,7 @@ header. See also the [macro overview page](../../features/macros.md). - [**JSON_HAS_STD_FORMAT**](json_has_std_format.md) - control `std::format`/`std::formatter` support - [**JSON_HAS_THREE_WAY_COMPARISON**](json_has_three_way_comparison.md) - control 3-way comparison support - [**JSON_NO_IO**](json_no_io.md) - switch off functions relying on certain C++ I/O headers +- [**JSON_NO_THREAD_LOCAL**](json_no_thread_local.md) - switch off the use of `thread_local` storage - [**JSON_SKIP_UNSUPPORTED_COMPILER_CHECK**](json_skip_unsupported_compiler_check.md) - do not warn about unsupported compilers - [**JSON_USE_GLOBAL_UDLS**](json_use_global_udls.md) - place user-defined string literals (UDLs) into the global namespace - [**JSON_USE_SIMDUTF**](json_use_simdutf.md) - use the simdutf library to accelerate UTF-8 validation diff --git a/docs/mkdocs/docs/api/macros/json_no_thread_local.md b/docs/mkdocs/docs/api/macros/json_no_thread_local.md new file mode 100644 index 000000000..126116ec3 --- /dev/null +++ b/docs/mkdocs/docs/api/macros/json_no_thread_local.md @@ -0,0 +1,47 @@ +# JSON_NO_THREAD_LOCAL + +```cpp +#define JSON_NO_THREAD_LOCAL +``` + +When defined, the library does not use `#!cpp thread_local` storage. This is relevant for the few environments whose +toolchain does not support it. + +The copy constructor copies the first levels of a value by copying the containers, which copy their elements, and +completes whatever is nested deeper than that without the call stack, so that copying a value cannot exhaust the stack +however deeply it is nested. It counts the levels it has descended into in a `#!cpp thread_local` variable, as a counter +shared between threads would be raced. + +Without that counter, no descent can be bounded safely, so objects and arrays are copied without the call stack right +away. Copying keeps working exactly as it does otherwise - the same values come out, and deeply nested values are copied +just as safely - but copying is slower, because the containers no longer copy themselves. Copying the benchmark +documents takes 9% (`canada.json`) to 34% (`twitter.json`) longer; values built mostly from objects are affected the +most. + +## Default definition + +By default, `#!cpp JSON_NO_THREAD_LOCAL` is not defined. + +```cpp +#undef JSON_NO_THREAD_LOCAL +``` + +The library defines it by itself for Clang targeting MinGW, which does not survive the `#!cpp thread_local` storage: +copying a value segfaults there, with both old and current Clang versions, while GCC targeting MinGW is unaffected. + +## Examples + +??? example + + The code below forces the library not to use `#!cpp thread_local` storage. + + ```cpp + #define JSON_NO_THREAD_LOCAL 1 + #include + + ... + ``` + +## Version history + +- Added in version 3.12.1. diff --git a/docs/mkdocs/docs/features/macros.md b/docs/mkdocs/docs/features/macros.md index 927e00df3..e7baba0ae 100644 --- a/docs/mkdocs/docs/features/macros.md +++ b/docs/mkdocs/docs/features/macros.md @@ -91,6 +91,13 @@ security reasons (e.g., Intel Software Guard Extensions (SGX)). See [full documentation of `JSON_NO_IO`](../api/macros/json_no_io.md). +## `JSON_NO_THREAD_LOCAL` + +When defined, the library does not use `#!cpp thread_local` storage. Copying a value then always avoids the call stack +rather than descending into a bounded number of levels first, which is slower but yields the same values. + +See [full documentation of `JSON_NO_THREAD_LOCAL`](../api/macros/json_no_thread_local.md). + ## `JSON_SKIP_LIBRARY_VERSION_CHECK` When defined, the library will not create a compiler warning when a different version of the library was already diff --git a/docs/mkdocs/mkdocs.yml b/docs/mkdocs/mkdocs.yml index 856d86e6d..8f92a838f 100644 --- a/docs/mkdocs/mkdocs.yml +++ b/docs/mkdocs/mkdocs.yml @@ -292,6 +292,7 @@ nav: - 'JSON_HAS_THREE_WAY_COMPARISON': api/macros/json_has_three_way_comparison.md - 'JSON_NOEXCEPTION': api/macros/json_noexception.md - 'JSON_NO_IO': api/macros/json_no_io.md + - 'JSON_NO_THREAD_LOCAL': api/macros/json_no_thread_local.md - 'JSON_SKIP_LIBRARY_VERSION_CHECK': api/macros/json_skip_library_version_check.md - 'JSON_SKIP_UNSUPPORTED_COMPILER_CHECK': api/macros/json_skip_unsupported_compiler_check.md - 'JSON_STRICT_NUL_HANDLING': api/macros/json_strict_nul_handling.md diff --git a/include/nlohmann/detail/macro_scope.hpp b/include/nlohmann/detail/macro_scope.hpp index 8aacc0c51..def9da6f8 100644 --- a/include/nlohmann/detail/macro_scope.hpp +++ b/include/nlohmann/detail/macro_scope.hpp @@ -186,6 +186,15 @@ #define JSON_NO_UNIQUE_ADDRESS #endif +// Clang targeting MinGW does not survive the thread_local storage the copy +// constructor uses to bound its descent: every test that copies a value +// segfaults with clang 11.0.1 and clang 18.1.8, while the same tests pass with +// GCC targeting MinGW and with every other toolchain the library is tested on. +// Copying works the same way without the counter, only more slowly. +#if !defined(JSON_NO_THREAD_LOCAL) && defined(__clang__) && defined(__MINGW32__) + #define JSON_NO_THREAD_LOCAL 1 +#endif + // disable documentation warnings on clang #if defined(__clang__) #pragma clang diagnostic push diff --git a/include/nlohmann/json.hpp b/include/nlohmann/json.hpp index a63363fbf..9c1d821e1 100644 --- a/include/nlohmann/json.hpp +++ b/include/nlohmann/json.hpp @@ -28,14 +28,14 @@ #pragma GCC diagnostic ignored "-Wignored-attributes" #endif -#include // all_of, find, for_each +#include // all_of, find, for_each, none_of #include // nullptr_t, ptrdiff_t, size_t #include // hash, less #include // initializer_list #ifndef JSON_NO_IO #include // istream, ostream #endif // JSON_NO_IO -#include // random_access_iterator_tag +#include // make_move_iterator, random_access_iterator_tag #include // unique_ptr #include // string, stoi, to_string #include // declval, forward, move, pair, swap @@ -896,6 +896,352 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec return j; } +#ifndef JSON_NO_THREAD_LOCAL + /// the number of levels an operation descends into before it finishes the + /// value below it without the call stack + static constexpr std::uint8_t nesting_depth_limit() + { + return 128; + } + + /*! + @brief how many levels the operation going on in this thread has descended into + + Copying a value and comparing two values share this count. The library never + nests one inside the other - copying a value does not compare one, and + comparing two values does not copy them - and where user code nests them + anyway, sharing the count only ends a descent sooner than it had to, which + costs a little speed and is never wrong. + + A byte is enough: the count never exceeds the limit by more than the single + level that notices the limit has been reached. + */ + static std::uint8_t& nesting_depth() noexcept + { + static thread_local std::uint8_t depth = 0; // NOLINT(misc-use-internal-linkage) + return depth; + } +#endif + + /*! + @brief counts one level of a bounded descent for as long as it runs, and + reports whether the descent was still within the limit when it began + + Looks the count up and tests it against the limit itself, rather than + leaving that to the caller: either way it is reached exactly once, so + there is nothing to be gained by making the caller do it. + + Does nothing and is never @ref okay without thread-local storage, where no + descent can be bounded at all: a caller that only descends while this says + it may always ends up finishing without the call stack, exactly as if every + value were nested past the limit. + */ + class nesting_depth_guard + { + public: + nesting_depth_guard() noexcept +#ifdef JSON_NO_THREAD_LOCAL + : m_okay(false) +#else + : m_okay(nesting_depth() < nesting_depth_limit()) +#endif + { +#ifndef JSON_NO_THREAD_LOCAL + ++nesting_depth(); +#endif + } + + ~nesting_depth_guard() + { +#ifndef JSON_NO_THREAD_LOCAL + --nesting_depth(); +#endif + } + + nesting_depth_guard(const nesting_depth_guard&) = delete; + nesting_depth_guard& operator=(const nesting_depth_guard&) = delete; + nesting_depth_guard(nesting_depth_guard&&) = delete; + nesting_depth_guard& operator=(nesting_depth_guard&&) = delete; + + bool okay() const noexcept + { + return m_okay; + } + + private: + bool m_okay; + }; + + /// an entry of the iterative deep copy's worklist: a structured value and + /// the value that is to become its copy + using copy_worklist_t = std::vector>; + + /// scratch space to build the key skeleton of an object copy in one go + using copy_scratch_t = std::vector>; + + /// @brief copy everything of @a src into @a dst but its type and value + static void copy_metadata(const basic_json& src, basic_json& dst) + { + // a custom base class is only required to be copy-constructible and + // move-assignable, so the copy has to go through a temporary + static_cast(dst) = json_base_class_t(static_cast(src)); + +#if JSON_DIAGNOSTIC_POSITIONS + dst.start_position = src.start_position; + dst.end_position = src.end_position; +#endif + } + + /*! + @brief copy the value of @a src into @a dst, which must not be structured + + Objects and arrays are left alone: creating those is the one thing the copy + constructor and @ref copy_shallow do differently from one another, and it is + the reason copying a value can descend at all. + */ + /// @note inlined on purpose: both callers have already told an object or an + /// array apart from the rest, and letting the compiler fold that test + /// into this switch is worth a few percent when copying a value made + /// mostly of numbers + JSON_HEDLEY_ALWAYS_INLINE + static void copy_leaf_value(const basic_json& src, basic_json& dst) + { + switch (src.m_data.m_type) + { + case value_t::string: + { + dst.m_data.m_value = *src.m_data.m_value.string; + break; + } + + case value_t::binary: + { + dst.m_data.m_value = *src.m_data.m_value.binary; + break; + } + + case value_t::boolean: + { + dst.m_data.m_value = src.m_data.m_value.boolean; + break; + } + + case value_t::number_integer: + { + dst.m_data.m_value = src.m_data.m_value.number_integer; + break; + } + + case value_t::number_unsigned: + { + dst.m_data.m_value = src.m_data.m_value.number_unsigned; + break; + } + + case value_t::number_float: + { + dst.m_data.m_value = src.m_data.m_value.number_float; + break; + } + + case value_t::object: + case value_t::array: + case value_t::null: + case value_t::discarded: + default: + break; + } + } + + /*! + @brief copy everything of @a src into the null value @a dst but the children + + Objects and arrays are not copied here; they are appended to @a worklist to + be created later by @ref copy_iteratively. Until that happens, @a dst remains + a null value, so that a partially built copy can be destroyed at any point + without ever violating the class invariants. + */ + static void copy_shallow(const basic_json& src, basic_json& dst, copy_worklist_t& worklist) + { + copy_metadata(src, dst); + + if (src.m_data.m_type == value_t::object || src.m_data.m_type == value_t::array) + { + // defer: dst stays a null value until its container exists + worklist.emplace_back(&src, &dst); + return; + } + + copy_leaf_value(src, dst); + + // only now that the value exists may the type be set: had the creation + // of the value thrown, dst would have been left as a valid null value + dst.m_data.m_type = src.m_data.m_type; + } + + /// @brief create the copy of the array @a src in @a dst + /// @note structured elements are appended to @a worklist instead + static void copy_array_level(const basic_json& src, basic_json& dst, copy_worklist_t& worklist) + { + const array_t& src_array = *src.m_data.m_value.array; + + // create all elements up front: growing the array afterwards could + // invalidate the pointers that are handed to the worklist; resize() + // rather than the fill constructor, because not every array type + // provides the latter (e.g., ones without a matching allocator-aware + // fill constructor) + dst.m_data.m_value.array = create(); + dst.m_data.m_value.array->resize(src_array.size()); + + auto dst_it = dst.m_data.m_value.array->begin(); + for (auto src_it = src_array.cbegin(); src_it != src_array.cend(); ++src_it, ++dst_it) + { + copy_shallow(*src_it, *dst_it, worklist); + } + } + + /// @brief create the copy of the object @a src in @a dst + /// @note structured values are appended to @a worklist instead + static void copy_object_level(const basic_json& src, basic_json& dst, + copy_worklist_t& worklist, copy_scratch_t& scratch) + { + const object_t& src_object = *src.m_data.m_value.object; + + // build the complete key skeleton and hand it to the object's range + // constructor: adding the keys one by one would be quadratic for object + // types that are backed by a vector, such as nlohmann::ordered_map + scratch.clear(); + scratch.reserve(src_object.size()); + for (const auto& element : src_object) + { + scratch.emplace_back(element.first, basic_json()); + } + + dst.m_data.m_value.object = create(std::make_move_iterator(scratch.begin()), + std::make_move_iterator(scratch.end())); + scratch.clear(); + + // pair every value of the copy with its counterpart in the original; + // both are enumerated in the same order for every object type with a + // deterministic order, so the lookup is only needed for exotic ones + auto src_it = src_object.cbegin(); + for (auto& element : *dst.m_data.m_value.object) + { + if (JSON_HEDLEY_LIKELY(src_it != src_object.cend() && src_it->first == element.first)) + { + copy_shallow(src_it->second, element.second, worklist); + ++src_it; + } + else + { + const auto found = src_object.find(element.first); + JSON_ASSERT(found != src_object.cend()); + copy_shallow(found->second, element.second, worklist); + } + } + } + + /*! + @brief deep-copy the object or array @a src into this value without recursing + + The values whose copy has not been created yet are kept on an explicit + worklist rather than on the call stack. This is only reached for values + nested deeper than @ref nesting_depth_limit levels, which is why it copies + every container by hand instead of letting the container do it: the fast + ways of doing so would descend into the elements and defeat the purpose. + */ + void copy_iteratively(const basic_json& src) + { + copy_worklist_t worklist; + copy_scratch_t scratch; + + const basic_json* src_value = &src; + basic_json* dst_value = this; + + for (;;) + { + if (src_value->m_data.m_type == value_t::array) + { + copy_array_level(*src_value, *dst_value, worklist); + } + else + { + copy_object_level(*src_value, *dst_value, worklist, scratch); + } + + // the container is complete and will not be modified again + dst_value->set_parents(); + + if (worklist.empty()) + { + break; + } + + const auto& next = worklist.back(); + src_value = next.first; + dst_value = next.second; + worklist.pop_back(); + + // the value stops being a null value exactly here + dst_value->m_data.m_type = src_value->m_data.m_type; + } + } + + /*! + @brief copy one level of the object or array @a src into this value + + The container copies its own elements, which is the fastest way to fill it. + Every element that is structured itself comes back to @ref copy_structured. + */ + void copy_level(const basic_json& src) + { + if (m_data.m_type == value_t::object) + { + m_data.m_value = *src.m_data.m_value.object; + } + else + { + m_data.m_value = *src.m_data.m_value.array; + } + + set_parents(); + } + + /*! + @brief deep-copy the object or array @a src into this value + + Copying a container copies its elements, so a value nested deeply enough + used to exhaust the call stack. The descent is bounded here: the first + @ref nesting_depth_limit levels are copied by the containers themselves, just + as they always were, and anything below that is copied without the call + stack by @ref copy_iteratively. Copying a value can therefore no longer + exhaust the stack, however deeply it is nested, just like destroying one + cannot since #1436. + + Nothing has to be scanned or built by hand to reach that: a value that is + not nested deeper than the limit - all but a vanishing minority - is copied + exactly as it was before, and this whole detour costs it one counter. + + @sa https://github.com/nlohmann/json/issues/5387 + */ + void copy_structured(const basic_json& src) + { + const nesting_depth_guard guard; + + if (JSON_HEDLEY_LIKELY(guard.okay())) + { + copy_level(src); + return; + } + + // Finish this value without descending any further. It is completed + // before this returns, so a copy made by a custom base class - or by + // anything else that runs while a copy is going on - is unaffected by + // the copy it is nested in. + copy_iteratively(src); + } + + public: ////////////////////////// // JSON parser callback // @@ -1275,60 +1621,15 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // check of passed value is valid other.assert_invariant(); - switch (m_data.m_type) + if (m_data.m_type == value_t::object || m_data.m_type == value_t::array) { - case value_t::object: - { - m_data.m_value = *other.m_data.m_value.object; - break; - } - - case value_t::array: - { - m_data.m_value = *other.m_data.m_value.array; - break; - } - - case value_t::string: - { - m_data.m_value = *other.m_data.m_value.string; - break; - } - - case value_t::boolean: - { - m_data.m_value = other.m_data.m_value.boolean; - break; - } - - case value_t::number_integer: - { - m_data.m_value = other.m_data.m_value.number_integer; - break; - } - - case value_t::number_unsigned: - { - m_data.m_value = other.m_data.m_value.number_unsigned; - break; - } - - case value_t::number_float: - { - m_data.m_value = other.m_data.m_value.number_float; - break; - } - - case value_t::binary: - { - m_data.m_value = *other.m_data.m_value.binary; - break; - } - - case value_t::null: - case value_t::discarded: - default: - break; + // copying the container directly would call this constructor again + // for every element, once per nesting level + copy_structured(other); + } + else + { + copy_leaf_value(other, *this); } set_parents(); diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 4418a6c19..81b0e09f6 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -28,14 +28,14 @@ #pragma GCC diagnostic ignored "-Wignored-attributes" #endif -#include // all_of, find, for_each +#include // all_of, find, for_each, none_of #include // nullptr_t, ptrdiff_t, size_t #include // hash, less #include // initializer_list #ifndef JSON_NO_IO #include // istream, ostream #endif // JSON_NO_IO -#include // random_access_iterator_tag +#include // make_move_iterator, random_access_iterator_tag #include // unique_ptr #include // string, stoi, to_string #include // declval, forward, move, pair, swap @@ -2564,6 +2564,15 @@ JSON_HEDLEY_DIAGNOSTIC_POP #define JSON_NO_UNIQUE_ADDRESS #endif +// Clang targeting MinGW does not survive the thread_local storage the copy +// constructor uses to bound its descent: every test that copies a value +// segfaults with clang 11.0.1 and clang 18.1.8, while the same tests pass with +// GCC targeting MinGW and with every other toolchain the library is tested on. +// Copying works the same way without the counter, only more slowly. +#if !defined(JSON_NO_THREAD_LOCAL) && defined(__clang__) && defined(__MINGW32__) + #define JSON_NO_THREAD_LOCAL 1 +#endif + // disable documentation warnings on clang #if defined(__clang__) #pragma clang diagnostic push @@ -25152,6 +25161,352 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec return j; } +#ifndef JSON_NO_THREAD_LOCAL + /// the number of levels an operation descends into before it finishes the + /// value below it without the call stack + static constexpr std::uint8_t nesting_depth_limit() + { + return 128; + } + + /*! + @brief how many levels the operation going on in this thread has descended into + + Copying a value and comparing two values share this count. The library never + nests one inside the other - copying a value does not compare one, and + comparing two values does not copy them - and where user code nests them + anyway, sharing the count only ends a descent sooner than it had to, which + costs a little speed and is never wrong. + + A byte is enough: the count never exceeds the limit by more than the single + level that notices the limit has been reached. + */ + static std::uint8_t& nesting_depth() noexcept + { + static thread_local std::uint8_t depth = 0; // NOLINT(misc-use-internal-linkage) + return depth; + } +#endif + + /*! + @brief counts one level of a bounded descent for as long as it runs, and + reports whether the descent was still within the limit when it began + + Looks the count up and tests it against the limit itself, rather than + leaving that to the caller: either way it is reached exactly once, so + there is nothing to be gained by making the caller do it. + + Does nothing and is never @ref okay without thread-local storage, where no + descent can be bounded at all: a caller that only descends while this says + it may always ends up finishing without the call stack, exactly as if every + value were nested past the limit. + */ + class nesting_depth_guard + { + public: + nesting_depth_guard() noexcept +#ifdef JSON_NO_THREAD_LOCAL + : m_okay(false) +#else + : m_okay(nesting_depth() < nesting_depth_limit()) +#endif + { +#ifndef JSON_NO_THREAD_LOCAL + ++nesting_depth(); +#endif + } + + ~nesting_depth_guard() + { +#ifndef JSON_NO_THREAD_LOCAL + --nesting_depth(); +#endif + } + + nesting_depth_guard(const nesting_depth_guard&) = delete; + nesting_depth_guard& operator=(const nesting_depth_guard&) = delete; + nesting_depth_guard(nesting_depth_guard&&) = delete; + nesting_depth_guard& operator=(nesting_depth_guard&&) = delete; + + bool okay() const noexcept + { + return m_okay; + } + + private: + bool m_okay; + }; + + /// an entry of the iterative deep copy's worklist: a structured value and + /// the value that is to become its copy + using copy_worklist_t = std::vector>; + + /// scratch space to build the key skeleton of an object copy in one go + using copy_scratch_t = std::vector>; + + /// @brief copy everything of @a src into @a dst but its type and value + static void copy_metadata(const basic_json& src, basic_json& dst) + { + // a custom base class is only required to be copy-constructible and + // move-assignable, so the copy has to go through a temporary + static_cast(dst) = json_base_class_t(static_cast(src)); + +#if JSON_DIAGNOSTIC_POSITIONS + dst.start_position = src.start_position; + dst.end_position = src.end_position; +#endif + } + + /*! + @brief copy the value of @a src into @a dst, which must not be structured + + Objects and arrays are left alone: creating those is the one thing the copy + constructor and @ref copy_shallow do differently from one another, and it is + the reason copying a value can descend at all. + */ + /// @note inlined on purpose: both callers have already told an object or an + /// array apart from the rest, and letting the compiler fold that test + /// into this switch is worth a few percent when copying a value made + /// mostly of numbers + JSON_HEDLEY_ALWAYS_INLINE + static void copy_leaf_value(const basic_json& src, basic_json& dst) + { + switch (src.m_data.m_type) + { + case value_t::string: + { + dst.m_data.m_value = *src.m_data.m_value.string; + break; + } + + case value_t::binary: + { + dst.m_data.m_value = *src.m_data.m_value.binary; + break; + } + + case value_t::boolean: + { + dst.m_data.m_value = src.m_data.m_value.boolean; + break; + } + + case value_t::number_integer: + { + dst.m_data.m_value = src.m_data.m_value.number_integer; + break; + } + + case value_t::number_unsigned: + { + dst.m_data.m_value = src.m_data.m_value.number_unsigned; + break; + } + + case value_t::number_float: + { + dst.m_data.m_value = src.m_data.m_value.number_float; + break; + } + + case value_t::object: + case value_t::array: + case value_t::null: + case value_t::discarded: + default: + break; + } + } + + /*! + @brief copy everything of @a src into the null value @a dst but the children + + Objects and arrays are not copied here; they are appended to @a worklist to + be created later by @ref copy_iteratively. Until that happens, @a dst remains + a null value, so that a partially built copy can be destroyed at any point + without ever violating the class invariants. + */ + static void copy_shallow(const basic_json& src, basic_json& dst, copy_worklist_t& worklist) + { + copy_metadata(src, dst); + + if (src.m_data.m_type == value_t::object || src.m_data.m_type == value_t::array) + { + // defer: dst stays a null value until its container exists + worklist.emplace_back(&src, &dst); + return; + } + + copy_leaf_value(src, dst); + + // only now that the value exists may the type be set: had the creation + // of the value thrown, dst would have been left as a valid null value + dst.m_data.m_type = src.m_data.m_type; + } + + /// @brief create the copy of the array @a src in @a dst + /// @note structured elements are appended to @a worklist instead + static void copy_array_level(const basic_json& src, basic_json& dst, copy_worklist_t& worklist) + { + const array_t& src_array = *src.m_data.m_value.array; + + // create all elements up front: growing the array afterwards could + // invalidate the pointers that are handed to the worklist; resize() + // rather than the fill constructor, because not every array type + // provides the latter (e.g., ones without a matching allocator-aware + // fill constructor) + dst.m_data.m_value.array = create(); + dst.m_data.m_value.array->resize(src_array.size()); + + auto dst_it = dst.m_data.m_value.array->begin(); + for (auto src_it = src_array.cbegin(); src_it != src_array.cend(); ++src_it, ++dst_it) + { + copy_shallow(*src_it, *dst_it, worklist); + } + } + + /// @brief create the copy of the object @a src in @a dst + /// @note structured values are appended to @a worklist instead + static void copy_object_level(const basic_json& src, basic_json& dst, + copy_worklist_t& worklist, copy_scratch_t& scratch) + { + const object_t& src_object = *src.m_data.m_value.object; + + // build the complete key skeleton and hand it to the object's range + // constructor: adding the keys one by one would be quadratic for object + // types that are backed by a vector, such as nlohmann::ordered_map + scratch.clear(); + scratch.reserve(src_object.size()); + for (const auto& element : src_object) + { + scratch.emplace_back(element.first, basic_json()); + } + + dst.m_data.m_value.object = create(std::make_move_iterator(scratch.begin()), + std::make_move_iterator(scratch.end())); + scratch.clear(); + + // pair every value of the copy with its counterpart in the original; + // both are enumerated in the same order for every object type with a + // deterministic order, so the lookup is only needed for exotic ones + auto src_it = src_object.cbegin(); + for (auto& element : *dst.m_data.m_value.object) + { + if (JSON_HEDLEY_LIKELY(src_it != src_object.cend() && src_it->first == element.first)) + { + copy_shallow(src_it->second, element.second, worklist); + ++src_it; + } + else + { + const auto found = src_object.find(element.first); + JSON_ASSERT(found != src_object.cend()); + copy_shallow(found->second, element.second, worklist); + } + } + } + + /*! + @brief deep-copy the object or array @a src into this value without recursing + + The values whose copy has not been created yet are kept on an explicit + worklist rather than on the call stack. This is only reached for values + nested deeper than @ref nesting_depth_limit levels, which is why it copies + every container by hand instead of letting the container do it: the fast + ways of doing so would descend into the elements and defeat the purpose. + */ + void copy_iteratively(const basic_json& src) + { + copy_worklist_t worklist; + copy_scratch_t scratch; + + const basic_json* src_value = &src; + basic_json* dst_value = this; + + for (;;) + { + if (src_value->m_data.m_type == value_t::array) + { + copy_array_level(*src_value, *dst_value, worklist); + } + else + { + copy_object_level(*src_value, *dst_value, worklist, scratch); + } + + // the container is complete and will not be modified again + dst_value->set_parents(); + + if (worklist.empty()) + { + break; + } + + const auto& next = worklist.back(); + src_value = next.first; + dst_value = next.second; + worklist.pop_back(); + + // the value stops being a null value exactly here + dst_value->m_data.m_type = src_value->m_data.m_type; + } + } + + /*! + @brief copy one level of the object or array @a src into this value + + The container copies its own elements, which is the fastest way to fill it. + Every element that is structured itself comes back to @ref copy_structured. + */ + void copy_level(const basic_json& src) + { + if (m_data.m_type == value_t::object) + { + m_data.m_value = *src.m_data.m_value.object; + } + else + { + m_data.m_value = *src.m_data.m_value.array; + } + + set_parents(); + } + + /*! + @brief deep-copy the object or array @a src into this value + + Copying a container copies its elements, so a value nested deeply enough + used to exhaust the call stack. The descent is bounded here: the first + @ref nesting_depth_limit levels are copied by the containers themselves, just + as they always were, and anything below that is copied without the call + stack by @ref copy_iteratively. Copying a value can therefore no longer + exhaust the stack, however deeply it is nested, just like destroying one + cannot since #1436. + + Nothing has to be scanned or built by hand to reach that: a value that is + not nested deeper than the limit - all but a vanishing minority - is copied + exactly as it was before, and this whole detour costs it one counter. + + @sa https://github.com/nlohmann/json/issues/5387 + */ + void copy_structured(const basic_json& src) + { + const nesting_depth_guard guard; + + if (JSON_HEDLEY_LIKELY(guard.okay())) + { + copy_level(src); + return; + } + + // Finish this value without descending any further. It is completed + // before this returns, so a copy made by a custom base class - or by + // anything else that runs while a copy is going on - is unaffected by + // the copy it is nested in. + copy_iteratively(src); + } + + public: ////////////////////////// // JSON parser callback // @@ -25531,60 +25886,15 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // check of passed value is valid other.assert_invariant(); - switch (m_data.m_type) + if (m_data.m_type == value_t::object || m_data.m_type == value_t::array) { - case value_t::object: - { - m_data.m_value = *other.m_data.m_value.object; - break; - } - - case value_t::array: - { - m_data.m_value = *other.m_data.m_value.array; - break; - } - - case value_t::string: - { - m_data.m_value = *other.m_data.m_value.string; - break; - } - - case value_t::boolean: - { - m_data.m_value = other.m_data.m_value.boolean; - break; - } - - case value_t::number_integer: - { - m_data.m_value = other.m_data.m_value.number_integer; - break; - } - - case value_t::number_unsigned: - { - m_data.m_value = other.m_data.m_value.number_unsigned; - break; - } - - case value_t::number_float: - { - m_data.m_value = other.m_data.m_value.number_float; - break; - } - - case value_t::binary: - { - m_data.m_value = *other.m_data.m_value.binary; - break; - } - - case value_t::null: - case value_t::discarded: - default: - break; + // copying the container directly would call this constructor again + // for every element, once per nesting level + copy_structured(other); + } + else + { + copy_leaf_value(other, *this); } set_parents(); diff --git a/tests/CMakeLists.txt b/tests/CMakeLists.txt index 86b4825d7..7a9bc5471 100644 --- a/tests/CMakeLists.txt +++ b/tests/CMakeLists.txt @@ -75,7 +75,12 @@ target_compile_options(test_main PUBLIC # is annotated JSON_HEDLEY_NO_RETURN (it always throws), which # makes MSVC flag the code following its call in binary_reader.hpp # as unreachable for that instantiation, in both Debug and Release - $<$:/W4;/wd4566;/wd4996;/wd4702> + # Disable warning C4503: decorated name length exceeded, name was truncated; the deep + # copy support added for #5387 pushes the mangled name of + # std::allocator_traits<...>::construct for the custom-base-class + # test's map type past VS2015's limit. The name is only used for + # debug info, so truncation does not affect the build. + $<$:/W4;/wd4566;/wd4996;/wd4702;/wd4503> # https://github.com/nlohmann/json/issues/1114 $<$:/bigobj> $<$:-Wa,-mbig-obj> diff --git a/tests/src/unit-allocator.cpp b/tests/src/unit-allocator.cpp index 2dbb746b0..5c7b4230f 100644 --- a/tests/src/unit-allocator.cpp +++ b/tests/src/unit-allocator.cpp @@ -216,6 +216,57 @@ TEST_CASE("controlled bad_alloc") CHECK_THROWS_AS(my_json(s), std::bad_alloc&); next_construct_fails = false; } + + SECTION("basic_json(const basic_json&) of a deeply nested value (#5387)") + { + // Copying a value nested deeper than the descent bound builds the + // copy from the top down: every value whose own copy has not been + // made yet stays a null value until it is. Failing an allocation + // part-way through is what proves such a half-built copy can still + // be destroyed. + // + // Which path the failure lands in depends on the build: the first + // allocation of a copy belongs to the outermost level, so here it + // is the descending one. Built with JSON_NO_THREAD_LOCAL - as the + // ci_test_no_thread_local target builds the whole suite - no + // descent is made at all and the very same failure lands in the + // iterative path instead, part-way through its worklist. + const auto check_deep_copy = [](bool objects) + { + CAPTURE(objects); + + next_construct_fails = false; + + // deeper than the 128 levels the copy constructor descends into + const std::size_t depth = 300; + + my_json j = 1; + for (std::size_t i = 0; i < depth; ++i) + { + if (objects) + { + my_json wrapper = my_json::object(); + wrapper["a"] = std::move(j); + j = std::move(wrapper); + } + else + { + j = my_json::array({std::move(j)}); + } + } + + // NOLINTNEXTLINE(performance-unnecessary-copy-initialization): the copy is what is tested + CHECK_NOTHROW(my_json(j)); + + next_construct_fails = true; + // NOLINTNEXTLINE(performance-unnecessary-copy-initialization): the copy is what is tested + CHECK_THROWS_AS(my_json(j), std::bad_alloc&); + next_construct_fails = false; + }; + + check_deep_copy(false); + check_deep_copy(true); + } } } diff --git a/tests/src/unit-diagnostic-positions.cpp b/tests/src/unit-diagnostic-positions.cpp index 4d2f50a98..d607e935c 100644 --- a/tests/src/unit-diagnostic-positions.cpp +++ b/tests/src/unit-diagnostic-positions.cpp @@ -75,6 +75,72 @@ TEST_CASE("Better diagnostics with positions") CHECK(j.end_pos() == root.size()); } + SECTION("copying keeps the positions of nested values (#5387)") + { + // Values nested deeper than the copy constructor's descent bound are + // copied without the call stack, on a path that has to carry the + // positions over itself; shallower ones copy their containers, which + // bring the positions along. Both sides of the bound are checked here. + const auto check_copy = [](std::size_t depth, bool objects) + { + CAPTURE(depth) + CAPTURE(objects) + + const std::string opening = objects ? R"({"a":)" : "["; + const std::string closing = objects ? "}" : "]"; + + std::string text; + for (std::size_t i = 0; i < depth; ++i) + { + text += opening; + } + text += "12"; + for (std::size_t i = 0; i < depth; ++i) + { + text += closing; + } + + const json original = json::parse(text); + const json copy(original); // NOLINT(performance-unnecessary-copy-initialization) + + const json* o = &original; + const json* c = © + for (std::size_t level = 0; level <= depth; ++level) + { + CAPTURE(level) + REQUIRE(c->start_pos() == o->start_pos()); + REQUIRE(c->end_pos() == o->end_pos()); + + if (level < depth) + { + o = objects ? &o->at("a") : &o->at(0); + c = objects ? &c->at("a") : &c->at(0); + } + } + }; + + const auto check_arrays = [&check_copy](std::size_t depth) + { + check_copy(depth, false); + }; + const auto check_objects = [&check_copy](std::size_t depth) + { + check_copy(depth, true); + }; + + check_arrays(1); + check_arrays(127); + check_arrays(128); + check_arrays(129); + check_arrays(300); + + check_objects(1); + check_objects(127); + check_objects(128); + check_objects(129); + check_objects(300); + } + SECTION("JSON patch add to primitive parent (#4292)") { // the JSON Patch "add" target /foo/bar/baz has a string parent diff --git a/tests/src/unit-diagnostics.cpp b/tests/src/unit-diagnostics.cpp index 135ecf9d6..389a7a3d7 100644 --- a/tests/src/unit-diagnostics.cpp +++ b/tests/src/unit-diagnostics.cpp @@ -274,6 +274,63 @@ TEST_CASE("Regression tests for extended diagnostics") CHECK(j1["string"] == "t"); } + SECTION("Regression test for issue #5387 - copying keeps the parents of nested values") + { + // A value nested deeper than the copy constructor's descent bound is + // copied without the call stack. Every container that path creates has + // to have the parents of its children set, or the JSON Pointer in the + // diagnostic is cut short. + const std::size_t depth = 300; + + SECTION("objects") + { + json j = "not a number"; + std::string pointer; + for (std::size_t i = 0; i < depth; ++i) + { + j = json{{"a", j}}; + pointer += "/a"; + } + + json const copy(j); // NOLINT(performance-unnecessary-copy-initialization) + + const json* inner = © + for (std::size_t i = 0; i < depth; ++i) + { + inner = &inner->at("a"); + } + + std::string const expected = "[json.exception.type_error.302] (" + pointer + ") type must be number, but is string"; + int i = 0; + CHECK_THROWS_WITH_AS(i = inner->get(), expected.c_str(), json::type_error); + CHECK(i == 0); + } + + SECTION("arrays") + { + json j = "not a number"; + std::string pointer; + for (std::size_t i = 0; i < depth; ++i) + { + j = json::array({j}); + pointer += "/0"; + } + + json const copy(j); // NOLINT(performance-unnecessary-copy-initialization) + + const json* inner = © + for (std::size_t i = 0; i < depth; ++i) + { + inner = &inner->at(0); + } + + std::string const expected = "[json.exception.type_error.302] (" + pointer + ") type must be number, but is string"; + int i = 0; + CHECK_THROWS_WITH_AS(i = inner->get(), expected.c_str(), json::type_error); + CHECK(i == 0); + } + } + SECTION("Regression test - swap(array_t&)/swap(object_t&) must update JSON_DIAGNOSTICS parent pointers") { // swap(array_t&) diff --git a/tests/src/unit-large_json.cpp b/tests/src/unit-large_json.cpp index 98d16e336..f10c8be41 100644 --- a/tests/src/unit-large_json.cpp +++ b/tests/src/unit-large_json.cpp @@ -12,6 +12,7 @@ using nlohmann::json; #include +#include TEST_CASE("tests on very large JSONs") { @@ -27,3 +28,153 @@ TEST_CASE("tests on very large JSONs") } } +namespace +{ + +// Descend a chain of single-element containers and return the value at its end, +// reporting the number of levels traversed in @a depth. +// +// The values in the test case below are nested far deeper than the call stack +// can follow, so they must not be inspected with operator== or dump(): both are +// still recursive and would overflow the stack themselves. +const json* innermost_value(const json& j, std::size_t& depth) +{ + const json* current = &j; + depth = 0; + + while ((current->is_array() || current->is_object()) && !current->empty()) + { + current = current->is_array() + ? ¤t->front() + : ¤t->begin().value(); + ++depth; + } + + return current; +} + +} // namespace + +TEST_CASE("tests on deeply nested JSONs") +{ + // deep enough to exhaust the call stack, but small enough to stay cheap: + // parsing is iterative, so building the values below costs little + const std::size_t depth = 100000; + + SECTION("issue #5387 - stack overflow in the copy constructor") + { + SECTION("array") + { + const json j = json::parse(std::string(depth, '[') + '0' + std::string(depth, ']')); + + const json copy(j); // NOLINT(performance-unnecessary-copy-initialization): the copy is what is tested + + std::size_t copy_depth = 0; + CHECK(*innermost_value(copy, copy_depth) == 0); + CHECK(copy_depth == depth); + } + + SECTION("object") + { + std::string s; + s.reserve((6 * depth) + 1); + for (std::size_t i = 0; i < depth; ++i) + { + s += "{\"a\":"; + } + s += '1'; + s.append(depth, '}'); + + const json j = json::parse(s); + + const json copy(j); // NOLINT(performance-unnecessary-copy-initialization): the copy is what is tested + + std::size_t copy_depth = 0; + CHECK(*innermost_value(copy, copy_depth) == 1); + CHECK(copy_depth == depth); + } + + SECTION("copy assignment") + { + // operator=(basic_json) takes its argument by value, so the deep + // copy happens in the copy constructor + const json j = json::parse(std::string(depth, '[') + '0' + std::string(depth, ']')); + + json target; + target = j; + + std::size_t target_depth = 0; + CHECK(*innermost_value(target, target_depth) == 0); + CHECK(target_depth == depth); + } + + SECTION("depths around the bound of the recursive descent") + { + // The copy constructor descends into a bounded number of levels and + // completes whatever is below that without the call stack. Cover + // every depth around that bound, so that the two ways of copying + // are known to meet cleanly - wherever the bound is set. + for (std::size_t d = 1; d <= 300; ++d) + { + CAPTURE(d); + + const json array = json::parse(std::string(d, '[') + '0' + std::string(d, ']')); + const json array_copy(array); // NOLINT(performance-unnecessary-copy-initialization): the copy is what is tested + std::size_t array_depth = 0; + CHECK(*innermost_value(array_copy, array_depth) == 0); + CHECK(array_depth == d); + + std::string object_text; + for (std::size_t i = 0; i < d; ++i) + { + object_text += "{\"a\":"; + } + object_text += '1'; + object_text.append(d, '}'); + + const json object = json::parse(object_text); + const json object_copy(object); // NOLINT(performance-unnecessary-copy-initialization): the copy is what is tested + std::size_t object_depth = 0; + CHECK(*innermost_value(object_copy, object_depth) == 1); + CHECK(object_depth == d); + } + } + + SECTION("a value that is deep in one place only") + { + json j = json::object(); + j["shallow"] = 1; + j["deep"] = json::parse(std::string(depth, '[') + '0' + std::string(depth, ']')); + j["also_shallow"] = json::array({1, 2, 3}); + + const json copy(j); + + CHECK(copy["shallow"] == 1); + CHECK(copy["also_shallow"] == json::array({1, 2, 3})); + + std::size_t deep_depth = 0; + CHECK(*innermost_value(copy["deep"], deep_depth) == 0); + CHECK(deep_depth == depth); + } + + SECTION("the copy is independent of the original") + { + const json j = json::parse(std::string(depth, '[') + '0' + std::string(depth, ']')); + + json copy(j); + + // reach the innermost value without recursing and replace it + json* current = © + while (current->is_array() && !current->empty()) + { + current = ¤t->front(); + } + *current = 42; + + std::size_t unused = 0; + CHECK(*innermost_value(copy, unused) == 42); + CHECK(*innermost_value(j, unused) == 0); + } + } +} + diff --git a/tests/src/unit-ordered_json.cpp b/tests/src/unit-ordered_json.cpp index 62a949a7f..45fbf5493 100644 --- a/tests/src/unit-ordered_json.cpp +++ b/tests/src/unit-ordered_json.cpp @@ -82,6 +82,40 @@ TEST_CASE("regression test for issue #3732 - iteration_proxy_value(fn); } +TEST_CASE("copying an ordered_json with nested values") +{ + // ordered_map is backed by a vector, so copying an object that has + // structured values takes a different route than copying a std::map-backed + // one; see https://github.com/nlohmann/json/issues/5387 + ordered_json oj; + oj["z"] = 1; + oj["a"]["y"] = 2; + oj["a"]["b"]["x"] = 3; + oj["m"] = {1, 2, {{"w", 4}}}; + + const ordered_json copy(oj); + + SECTION("the copy is equal to the original") + { + CHECK(copy == oj); + CHECK(copy.dump() == oj.dump()); + } + + SECTION("the key order is preserved at every level") + { + CHECK(copy.dump() == R"({"z":1,"a":{"y":2,"b":{"x":3}},"m":[1,2,{"w":4}]})"); + } + + SECTION("the copy is independent of the original") + { + ordered_json mutated(oj); + mutated["a"]["b"]["x"] = 99; + + CHECK(oj["a"]["b"]["x"] == 3); + CHECK(mutated["a"]["b"]["x"] == 99); + } +} + TEST_CASE("regression test - diff() must account for ordered_json member order") { SECTION("pure reorder, no value changes") From f3768d686841e0417f7ca09f0d98d58e08ad7e0b Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Thu, 24 Sep 2026 17:01:19 +0200 Subject: [PATCH 23/64] Match ABI tag order in namespace tests to abi_macros.hpp (#5551) * Match ABI tag order in namespace tests to abi_macros.hpp NLOHMANN_JSON_ABI_TAGS concatenates the tags as _diag, _ldvcmp, _dp, but the default and noversion ABI tests expected _diag, _dp, _ldvcmp. The tests therefore failed whenever both JSON_DIAGNOSTIC_POSITIONS and JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON were enabled, a combination CI never exercises. Reorder the expectations to match the header. Also document the _dp tag in the namespace feature page, which listed only _diag and _ldvcmp. Signed-off-by: Niels Lohmann * Test the ABI namespace with all ABI tags enabled Build the default and noversion ABI config tests a second time with JSON_DIAGNOSTICS, JSON_DIAGNOSTIC_POSITIONS and JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON all set, so the expected tag order is checked on every test run instead of depending on which CMake options a CI job happens to enable. Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- docs/mkdocs/docs/features/namespace.md | 1 + tests/abi/config/CMakeLists.txt | 14 ++++++++++++++ tests/abi/config/default.cpp | 8 ++++---- tests/abi/config/noversion.cpp | 8 ++++---- 4 files changed, 23 insertions(+), 8 deletions(-) diff --git a/docs/mkdocs/docs/features/namespace.md b/docs/mkdocs/docs/features/namespace.md index 5542c1f88..c4efe772a 100644 --- a/docs/mkdocs/docs/features/namespace.md +++ b/docs/mkdocs/docs/features/namespace.md @@ -15,6 +15,7 @@ The complete default namespace name is derived as follows: - [`JSON_DIAGNOSTICS`](../api/macros/json_diagnostics.md) defined non-zero appends `_diag`. - [`JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON`](../api/macros/json_use_legacy_discarded_value_comparison.md) defined non-zero appends `_ldvcmp`. + - [`JSON_DIAGNOSTIC_POSITIONS`](../api/macros/json_diagnostic_positions.md) defined non-zero appends `_dp`. - The inline namespace ends with the suffix `_v` followed by the 3 components of the version number separated by underscores. To omit the version component, see [Disabling the version component](#disabling-the-version-component) below. diff --git a/tests/abi/config/CMakeLists.txt b/tests/abi/config/CMakeLists.txt index 3a8367690..52941dc33 100644 --- a/tests/abi/config/CMakeLists.txt +++ b/tests/abi/config/CMakeLists.txt @@ -14,6 +14,20 @@ add_test( NAME test-abi_config_noversion COMMAND abi_config_noversion ${DOCTEST_TEST_FILTER}) +# test default and no version namespace with all ABI tags enabled, so the +# expected tag order is checked regardless of the JSON_* CMake options +foreach(test default noversion) + add_executable(abi_config_${test}_all_tags ${test}.cpp) + target_compile_definitions(abi_config_${test}_all_tags PRIVATE + JSON_DIAGNOSTICS=1 + JSON_DIAGNOSTIC_POSITIONS=1 + JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON=1) + target_link_libraries(abi_config_${test}_all_tags PRIVATE abi_compat_main) + add_test( + NAME test-abi_config_${test}_all_tags + COMMAND abi_config_${test}_all_tags ${DOCTEST_TEST_FILTER}) +endforeach() + # test custom namespace add_executable(abi_config_custom custom.cpp) target_link_libraries(abi_config_custom PRIVATE abi_compat_main) diff --git a/tests/abi/config/default.cpp b/tests/abi/config/default.cpp index 0edc12e62..f3ee23110 100644 --- a/tests/abi/config/default.cpp +++ b/tests/abi/config/default.cpp @@ -24,14 +24,14 @@ TEST_CASE("default namespace") expected += "_diag"; #endif -#if JSON_DIAGNOSTIC_POSITIONS - expected += "_dp"; -#endif - #if JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON expected += "_ldvcmp"; #endif +#if JSON_DIAGNOSTIC_POSITIONS + expected += "_dp"; +#endif + expected += "_v" STRINGIZE(NLOHMANN_JSON_VERSION_MAJOR); expected += "_" STRINGIZE(NLOHMANN_JSON_VERSION_MINOR); expected += "_" STRINGIZE(NLOHMANN_JSON_VERSION_PATCH) "::basic_json"; diff --git a/tests/abi/config/noversion.cpp b/tests/abi/config/noversion.cpp index 2ae5cf5ac..cbdcb149b 100644 --- a/tests/abi/config/noversion.cpp +++ b/tests/abi/config/noversion.cpp @@ -25,14 +25,14 @@ TEST_CASE("default namespace without version component") expected += "_diag"; #endif -#if JSON_DIAGNOSTIC_POSITIONS - expected += "_dp"; -#endif - #if JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON expected += "_ldvcmp"; #endif +#if JSON_DIAGNOSTIC_POSITIONS + expected += "_dp"; +#endif + expected += "::basic_json"; // fallback for Clang From 918da646577fd1a004cd72efbc7cbce2f578783b Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Thu, 24 Sep 2026 17:01:31 +0200 Subject: [PATCH 24/64] Keep BJData ndarray annotations that would not survive a round trip as objects (#5542) write_bjdata_ndarray() encoded a JData-annotated object as a BJData ND-array whenever its dimensions' product matched _ArrayData_.size(), which lost information in two ways: - _ArrayData_ was never required to be an array. null has size 0, any other scalar has size 1, and iterating an object visits its values, so e.g. {"_ArraySize_":[1],"_ArrayData_":5} was written as the array [5], and an object _ArrayData_ came back as an array. - The reader only restores an annotated object from an ND-array with at least two non-zero dimensions that is not a 1xN row vector; an empty, 1-D, row-vector, or zero-sized shape is read back as a plain array. The writer nonetheless emitted ND-array headers for these shapes, so the annotation was silently dropped. OSS-Fuzz issue 563659413 hit this in parse_bjdata_fuzzer: an empty binary _ArraySize_ is written as a plain object and read back as an empty array, after which {"_ArrayType_":"int16","_ArraySize_":[],"_ArrayData_":null} was encoded as the ND-array header "[$I#[]" and re-read as [], failing the harness's value-stability check. Such objects now fall back to a plain object encoding, which round-trips. Genuine ND-arrays (two or more positive dimensions, not a 1xN row vector) are encoded exactly as before. Existing fallback tests that used 1-D shapes are moved to 2-D shapes so they keep exercising the check they were written for, and the BJData documentation is updated. Signed-off-by: Niels Lohmann --- .../docs/features/binary_formats/bjdata.md | 18 +-- .../nlohmann/detail/output/binary_writer.hpp | 30 ++++- single_include/nlohmann/json.hpp | 30 ++++- tests/src/unit-bjdata.cpp | 112 +++++++++++++++--- 4 files changed, 156 insertions(+), 34 deletions(-) diff --git a/docs/mkdocs/docs/features/binary_formats/bjdata.md b/docs/mkdocs/docs/features/binary_formats/bjdata.md index d3f63a9b8..4cb61053f 100644 --- a/docs/mkdocs/docs/features/binary_formats/bjdata.md +++ b/docs/mkdocs/docs/features/binary_formats/bjdata.md @@ -116,18 +116,22 @@ The library uses the following mapping from JSON values types to BJData types ac ``` Likewise, when a JSON object in the above form is serialized using - [`to_bjdata`](../../api/basic_json/to_bjdata.md), it is automatically converted into a compact BJData ND-array. When - the 1-dimensional vector stored in `"_ArraySize_"` contains a single integer or two integers with one being 1, a - regular 1-D optimized array is generated instead. + [`to_bjdata`](../../api/basic_json/to_bjdata.md), it is automatically converted into a compact BJData ND-array. - An object is only converted if the annotation actually describes a packed array; otherwise it is serialized as a - regular JSON object. This requires all of the following: + When parsing, an ND-array whose dimension vector is empty, contains a single integer, contains two integers with the + first being 1, or contains a 0 is returned as a regular (possibly empty) array rather than an annotated object. + + An object is only converted if the annotation describes a packed array that is parsed back into the same annotated + object; otherwise it is serialized as a regular JSON object, so the annotation is never lost in a round trip. This requires + all of the following: - `"_ArrayType_"` is one of `uint8`, `int8`, `uint16`, `int16`, `uint32`, `int32`, `uint64`, `int64`, `single`, `double`, `char`, or `byte`, - `"_ArraySize_"` is an array, since the dimensions are written as the ND-array header's length, - - every entry of `"_ArraySize_"` is a non-negative integer, and their product is representable as a `std::size_t`, - - `"_ArrayData_"` holds exactly that many elements, and + - `"_ArraySize_"` has at least two entries and is not a 1×N row vector (first entry 1), since other shapes are + parsed back as a regular array, + - every entry of `"_ArraySize_"` is a positive integer, and their product is representable as a `std::size_t`, + - `"_ArrayData_"` is an array holding exactly that many elements, and - every element of `"_ArrayData_"` is a number of the kind named by `"_ArrayType_"` (a floating-point number for `single` and `double`, an integer otherwise). diff --git a/include/nlohmann/detail/output/binary_writer.hpp b/include/nlohmann/detail/output/binary_writer.hpp index a355f1c15..495f872c0 100644 --- a/include/nlohmann/detail/output/binary_writer.hpp +++ b/include/nlohmann/detail/output/binary_writer.hpp @@ -1800,8 +1800,19 @@ class binary_writer return true; } - std::size_t len = (value.at(key).empty() ? 0 : 1); - for (const auto& el : value.at(key)) + // the reader only restores an annotated object from an ND-array header + // with at least two dimensions: an empty dimension vector, a single + // dimension, or a 1xN row vector is read back as a plain array, which + // would silently drop the annotation, so such an object falls back to + // a plain object encoding instead + const auto& dims = value.at(key); + if (dims.size() < 2 || (dims.size() == 2 && dims.at(0).is_number_integer() && dims.at(0).template get() == 1)) + { + return true; + } + + std::size_t len = 1; + for (const auto& el : dims) { // a dimension is read as an unsigned value below, so anything that // is not a non-negative integer is rejected: a non-integer entry @@ -1823,15 +1834,26 @@ class binary_writer return true; } const auto dim_size = static_cast(dim); - if (dim_size != 0 && len > (std::numeric_limits::max)() / dim_size) + + // the reader turns an ND-array with any zero dimension into an + // empty plain array, dropping the annotation, so keep the object + if (dim_size == 0) + { + return true; + } + if (len > (std::numeric_limits::max)() / dim_size) { return true; } len *= dim_size; } + // the elements are written from _ArrayData_ as a flat list, so it has + // to be an array: size() is 0 for null and 1 for any other scalar, and + // iterating an object visits its values, so any of these could match + // the dimensions by accident and be encoded as an unrelated ND-array key = "_ArrayData_"; - if (value.at(key).size() != len) + if (!value.at(key).is_array() || value.at(key).size() != len) { return true; } diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 81b0e09f6..1d5e21927 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -20664,8 +20664,19 @@ class binary_writer return true; } - std::size_t len = (value.at(key).empty() ? 0 : 1); - for (const auto& el : value.at(key)) + // the reader only restores an annotated object from an ND-array header + // with at least two dimensions: an empty dimension vector, a single + // dimension, or a 1xN row vector is read back as a plain array, which + // would silently drop the annotation, so such an object falls back to + // a plain object encoding instead + const auto& dims = value.at(key); + if (dims.size() < 2 || (dims.size() == 2 && dims.at(0).is_number_integer() && dims.at(0).template get() == 1)) + { + return true; + } + + std::size_t len = 1; + for (const auto& el : dims) { // a dimension is read as an unsigned value below, so anything that // is not a non-negative integer is rejected: a non-integer entry @@ -20687,15 +20698,26 @@ class binary_writer return true; } const auto dim_size = static_cast(dim); - if (dim_size != 0 && len > (std::numeric_limits::max)() / dim_size) + + // the reader turns an ND-array with any zero dimension into an + // empty plain array, dropping the annotation, so keep the object + if (dim_size == 0) + { + return true; + } + if (len > (std::numeric_limits::max)() / dim_size) { return true; } len *= dim_size; } + // the elements are written from _ArrayData_ as a flat list, so it has + // to be an array: size() is 0 for null and 1 for any other scalar, and + // iterating an object visits its values, so any of these could match + // the dimensions by accident and be encoded as an unrelated ND-array key = "_ArrayData_"; - if (value.at(key).size() != len) + if (!value.at(key).is_array() || value.at(key).size() != len) { return true; } diff --git a/tests/src/unit-bjdata.cpp b/tests/src/unit-bjdata.cpp index 334259fb7..9bba141d2 100644 --- a/tests/src/unit-bjdata.cpp +++ b/tests/src/unit-bjdata.cpp @@ -2604,25 +2604,25 @@ TEST_CASE("BJData") // that still round-trips. // string data declared as a uint64 array - json const j_str = json({{"_ArrayType_", "uint64"}, {"_ArraySize_", {1}}, {"_ArrayData_", {"pointer"}}}); + json const j_str = json({{"_ArrayType_", "uint64"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {"pointer", "value"}}}); const auto out_str = json::to_bjdata(j_str); CHECK(out_str.at(0) == '{'); CHECK(json::from_bjdata(out_str) == j_str); // integer data declared as a double array - json const j_float = json({{"_ArrayType_", "double"}, {"_ArraySize_", {2}}, {"_ArrayData_", {1, 2}}}); + json const j_float = json({{"_ArrayType_", "double"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {1, 2}}}); const auto out_float = json::to_bjdata(j_float); CHECK(out_float.at(0) == '{'); CHECK(json::from_bjdata(out_float) == j_float); // a non-integer shape entry is likewise not treated as an ndarray - json const j_size = json({{"_ArrayType_", "uint8"}, {"_ArraySize_", {"x"}}, {"_ArrayData_", {1}}}); + json const j_size = json({{"_ArrayType_", "uint8"}, {"_ArraySize_", {"x", 1}}, {"_ArrayData_", {1}}}); const auto out_size = json::to_bjdata(j_size); CHECK(out_size.at(0) == '{'); CHECK(json::from_bjdata(out_size) == j_size); // a negative shape entry is not a usable dimension either - json const j_neg = json::parse(R"({"_ArrayType_":"uint8","_ArraySize_":[-1],"_ArrayData_":[1]})"); + json const j_neg = json::parse(R"({"_ArrayType_":"uint8","_ArraySize_":[-1,1],"_ArrayData_":[1]})"); const auto out_neg = json::to_bjdata(j_neg); CHECK(out_neg.at(0) == '{'); CHECK(json::from_bjdata(out_neg) == j_neg); @@ -2657,14 +2657,14 @@ TEST_CASE("BJData") } // negative values under a signed type behave the same way - const auto from_neg = json::to_bjdata(json::parse(R"({"_ArrayType_":"int32","_ArraySize_":[2],"_ArrayData_":[-5,7]})")); + const auto from_neg = json::to_bjdata(json::parse(R"({"_ArrayType_":"int32","_ArraySize_":[2,1],"_ArrayData_":[-5,7]})")); CHECK(from_neg.at(0) == '['); - CHECK(from_neg == json::to_bjdata(json({{"_ArrayType_", "int32"}, {"_ArraySize_", {2}}, {"_ArrayData_", {-5, 7}}}))); + CHECK(from_neg == json::to_bjdata(json({{"_ArrayType_", "int32"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {-5, 7}}}))); // and so do the floating point types - const auto from_float = json::to_bjdata(json::parse(R"({"_ArrayType_":"double","_ArraySize_":[2],"_ArrayData_":[1.5,2.5]})")); + const auto from_float = json::to_bjdata(json::parse(R"({"_ArrayType_":"double","_ArraySize_":[2,1],"_ArrayData_":[1.5,2.5]})")); CHECK(from_float.at(0) == '['); - CHECK(from_float == json::to_bjdata(json({{"_ArrayType_", "double"}, {"_ArraySize_", {2}}, {"_ArrayData_", {1.5, 2.5}}}))); + CHECK(from_float == json::to_bjdata(json({{"_ArrayType_", "double"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {1.5, 2.5}}}))); } SECTION("optimized ndarray (type and vector-size as 1D array)") @@ -2835,7 +2835,7 @@ TEST_CASE("BJData") // a single dimension that does not fit into std::size_t is // rejected for the same reason (only observable where // std::size_t is narrower than 64 bit) - json j_huge = json({{"_ArrayData_", json::array()}, {"_ArraySize_", {18446744073709551615ull}}, {"_ArrayType_", "uint8"}}); + json j_huge = json({{"_ArrayData_", json::array()}, {"_ArraySize_", {18446744073709551615ull, 2}}, {"_ArrayType_", "uint8"}}); CHECK(json::from_bjdata(json::to_bjdata(j_huge), true, true) == j_huge); // a well-formed ndarray is still encoded as one @@ -2878,42 +2878,116 @@ TEST_CASE("BJData") // object encoding that still round-trips (see GitHub issue #5403) // an unsigned element that does not fit uint8 - json const j_uint8 = json({{"_ArrayType_", "uint8"}, {"_ArraySize_", {2}}, {"_ArrayData_", {1, 256}}}); + json const j_uint8 = json({{"_ArrayType_", "uint8"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {1, 256}}}); const auto out_uint8 = json::to_bjdata(j_uint8); CHECK(out_uint8.at(0) == '{'); CHECK(json::from_bjdata(out_uint8) == j_uint8); // a signed element that does not fit int8 - json const j_int8 = json({{"_ArrayType_", "int8"}, {"_ArraySize_", {2}}, {"_ArrayData_", {1, 200}}}); + json const j_int8 = json({{"_ArrayType_", "int8"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {1, 200}}}); const auto out_int8 = json::to_bjdata(j_int8); CHECK(out_int8.at(0) == '{'); CHECK(json::from_bjdata(out_int8) == j_int8); // a negative element is likewise out of range for an // unsigned _ArrayType_ - json const j_uint16_neg = json({{"_ArrayType_", "uint16"}, {"_ArraySize_", {2}}, {"_ArrayData_", {1, -1}}}); + json const j_uint16_neg = json({{"_ArrayType_", "uint16"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {1, -1}}}); const auto out_uint16_neg = json::to_bjdata(j_uint16_neg); CHECK(out_uint16_neg.at(0) == '{'); CHECK(json::from_bjdata(out_uint16_neg) == j_uint16_neg); // a double element that overflows to infinity when narrowed // to the "single" (float) precision named by _ArrayType_ - json const j_single = json({{"_ArrayType_", "single"}, {"_ArraySize_", {2}}, {"_ArrayData_", {1.5, 1e40}}}); + json const j_single = json({{"_ArrayType_", "single"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {1.5, 1e40}}}); const auto out_single = json::to_bjdata(j_single); CHECK(out_single.at(0) == '{'); CHECK(json::from_bjdata(out_single) == j_single); // in-range boundary values still use the compact ndarray encoding - json const j_uint8_ok = json({{"_ArrayType_", "uint8"}, {"_ArraySize_", {2}}, {"_ArrayData_", {0, 255}}}); - CHECK(json::to_bjdata(j_uint8_ok) == std::vector({'[', '$', 'U', '#', '[', 'i', 2, ']', 0, 255})); + json const j_uint8_ok = json({{"_ArrayType_", "uint8"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {0, 255}}}); + CHECK(json::to_bjdata(j_uint8_ok) == std::vector({'[', '$', 'U', '#', '[', 'i', 2, 'i', 1, ']', 0, 255})); - json const j_int8_ok = json({{"_ArrayType_", "int8"}, {"_ArraySize_", {2}}, {"_ArrayData_", {-128, 127}}}); - CHECK(json::to_bjdata(j_int8_ok) == std::vector({'[', '$', 'i', '#', '[', 'i', 2, ']', 0x80, 0x7F})); + json const j_int8_ok = json({{"_ArrayType_", "int8"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {-128, 127}}}); + CHECK(json::to_bjdata(j_int8_ok) == std::vector({'[', '$', 'i', '#', '[', 'i', 2, 'i', 1, ']', 0x80, 0x7F})); - json const j_single_ok = json({{"_ArrayType_", "single"}, {"_ArraySize_", {1}}, {"_ArrayData_", {1.5}}}); + json const j_single_ok = json({{"_ArrayType_", "single"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {1.5, -1.5}}}); const auto out_single_ok = json::to_bjdata(j_single_ok); CHECK(out_single_ok.at(0) == '['); - CHECK(json::from_bjdata(out_single_ok) == json({1.5f})); + CHECK(json::from_bjdata(out_single_ok) == json({{"_ArrayType_", "single"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {1.5f, -1.5f}}})); + } + + SECTION("ndarray that would not be read back as an annotated object stays as object") + { + // the reader only restores an annotated object from an ND-array + // with at least two non-zero dimensions that is not a 1xN row + // vector; any other shape is read back as a plain array. Writing + // such an object as an ND-array would drop its annotation, so it + // falls back to a plain object encoding that round-trips. + for (const char* text : + { + R"({"_ArrayType_":"int16","_ArraySize_":[],"_ArrayData_":[]})", + R"({"_ArrayType_":"int16","_ArraySize_":[2],"_ArrayData_":[1,2]})", + R"({"_ArrayType_":"int16","_ArraySize_":[1,2],"_ArrayData_":[1,2]})", + R"({"_ArrayType_":"int16","_ArraySize_":[0],"_ArrayData_":[]})", + R"({"_ArrayType_":"int16","_ArraySize_":[2,0],"_ArrayData_":[]})", + R"({"_ArrayType_":"int16","_ArraySize_":[0,2],"_ArrayData_":[]})" + }) + { + CAPTURE(text); + const json j = json::parse(text); + for (const bool use_size : + { + false, true + }) + { + const auto out = json::to_bjdata(j, use_size, use_size); + CHECK(out.at(0) == '{'); + CHECK(json::from_bjdata(out) == j); + } + } + + // a genuine ND-array still uses the compact encoding and round-trips + const json j_2d = json::parse(R"({"_ArrayType_":"int16","_ArraySize_":[2,1],"_ArrayData_":[1,2]})"); + const auto out_2d = json::to_bjdata(j_2d); + CHECK(out_2d.at(0) == '['); + CHECK(json::from_bjdata(out_2d) == j_2d); + } + + SECTION("ndarray with non-array _ArrayData_ stays as object") + { + // the elements are written from _ArrayData_ as a flat list, so it + // has to be an array: null has size 0, any other scalar has size 1, + // and iterating an object visits its values, so each of these could + // match the dimensions and be encoded as an unrelated ND-array + for (const char* text : + { + R"({"_ArrayType_":"int16","_ArraySize_":[2,1],"_ArrayData_":null})", + R"({"_ArrayType_":"int16","_ArraySize_":[2,1],"_ArrayData_":{"a":1,"b":2}})", + R"({"_ArrayType_":"int16","_ArraySize_":[1],"_ArrayData_":5})", + R"({"_ArrayType_":"int16","_ArraySize_":[],"_ArrayData_":null})" + }) + { + CAPTURE(text); + const json j = json::parse(text); + const auto out = json::to_bjdata(j); + CHECK(out.at(0) == '{'); + CHECK(json::from_bjdata(out) == j); + } + + // OSS-Fuzz issue 563659413: an empty binary _ArraySize_ is written + // as a plain object and read back as an empty array, after which + // the object with a null _ArrayData_ was encoded as an empty + // ND-array and re-read as [], so a second round trip lost the value + const std::vector input = + { + '{', 'U', 11, '_', 'A', 'r', 'r', 'a', 'y', 'D', 'a', 't', 'a', '_', 'Z', + 'U', 11, '_', 'A', 'r', 'r', 'a', 'y', 'T', 'y', 'p', 'e', '_', 'S', 'i', 5, 'i', 'n', 't', '1', '6', + 'U', 11, '_', 'A', 'r', 'r', 'a', 'y', 'S', 'i', 'z', 'e', '_', '[', '$', 'B', '#', '[', ']', '}' + }; + const json j1 = json::from_bjdata(input); + const json j2 = json::from_bjdata(json::to_bjdata(j1, false, false)); + CHECK(j2 == json::parse(R"({"_ArrayType_":"int16","_ArraySize_":[],"_ArrayData_":null})")); + CHECK(json::from_bjdata(json::to_bjdata(j2, false, false)) == j2); } SECTION("ndarray with _ArrayType_ \"byte\" is gated by the BJData draft version") From bc066c183810f2d8368b3675331a1e6c82165eea Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Thu, 24 Sep 2026 17:02:39 +0200 Subject: [PATCH 25/64] Make serve_header.py listen on localhost and limit its CORS header (#5564) Without a bind address in serve_header.yml, the server listened on all interfaces, so any machine on the network could fetch the header and trigger make runs in the working trees. It now listens on localhost unless configured otherwise; bind: null restores the old behavior. The header was also sent with Access-Control-Allow-Origin: *, letting any web page read it. CORS is only needed because Compiler Explorer downloads #include headers in the browser, so the header now goes only to https://godbolt.org and https://compiler-explorer.com, configurable with cors_origins. Signed-off-by: Niels Lohmann --- tools/serve_header/README.md | 4 ++++ tools/serve_header/serve_header.py | 24 +++++++++++++++++---- tools/serve_header/serve_header.yml.example | 11 ++++++++-- 3 files changed, 33 insertions(+), 6 deletions(-) diff --git a/tools/serve_header/README.md b/tools/serve_header/README.md index 0d0ed69f6..bdf2c60ec 100644 --- a/tools/serve_header/README.md +++ b/tools/serve_header/README.md @@ -60,6 +60,10 @@ int main() { `serve_header.py` will try to read a configuration file `serve_header.yml` in the top level or project root directory, and will fall back on built-in defaults if the file cannot be read. An annotated example configuration can be found in `tools/serve_header/serve_header.yml.example`. +By default, the server listens on `localhost` only, and only web pages from Compiler Explorer (`https://godbolt.org` and `https://compiler-explorer.com`) may read the header. +Set `bind` to serve other machines as well; anyone who can reach the server can then trigger `make` runs in your working trees. +Set `cors_origins` to allow other web pages. + ## Serving `json.hpp` from multiple project directory instances or working trees `serve_header.py` was designed with the goal of supporting multiple project roots or working trees at the same time. diff --git a/tools/serve_header/serve_header.py b/tools/serve_header/serve_header.py index e2da2dad0..1f29cb589 100755 --- a/tools/serve_header/serve_header.py +++ b/tools/serve_header/serve_header.py @@ -26,6 +26,10 @@ HEADER = 'json.hpp' DATETIME_FORMAT = '%Y-%m-%d %H:%M:%S' +# origins whose pages may read the served header from a browser; Compiler +# Explorer downloads #include headers client-side +DEFAULT_CORS_ORIGINS = ['https://godbolt.org', 'https://compiler-explorer.com'] + JSON_VERSION_RE = re.compile(r'\s*#\s*define\s+NLOHMANN_JSON_VERSION_MAJOR\s+') class ExitHandler(logging.StreamHandler): @@ -247,6 +251,8 @@ class WorkTrees(FileSystemEventHandler): self.observer.join() class HeaderRequestHandler(SimpleHTTPRequestHandler): # lgtm[py/missing-call-to-init] + cors_origins = DEFAULT_CORS_ORIGINS + def __init__(self, request, client_address, server): """.""" self.worktrees = server.worktrees @@ -310,8 +316,11 @@ class HeaderRequestHandler(SimpleHTTPRequestHandler): # lgtm[py/missing-call-to- # set content length super().send_header('Content-Length', length) - # CORS header - self.send_header('Access-Control-Allow-Origin', '*') + # CORS header; only for the configured origins + origin = self.headers.get('Origin') + if origin in self.cors_origins: + self.send_header('Access-Control-Allow-Origin', origin) + self.send_header('Vary', 'Origin') # prevent caching self.send_header('Cache-Control', 'no-cache, no-store, must-revalidate') self.send_header('Pragma', 'no-cache') @@ -383,8 +392,15 @@ if __name__ == '__main__': # find and monitor working trees worktrees = WorkTrees(config.get('root', '.')) - # start web server - infos = socket.getaddrinfo(config.get('bind', None), config.get('port', 8443), + # origins allowed to read the header from a browser + cors_origins = config.get('cors_origins', DEFAULT_CORS_ORIGINS) + if isinstance(cors_origins, str): + cors_origins = [cors_origins] + HeaderRequestHandler.cors_origins = cors_origins + + # start web server; only reachable from this machine unless configured + # otherwise (bind: null listens on all interfaces) + infos = socket.getaddrinfo(config.get('bind', 'localhost'), config.get('port', 8443), type=socket.SOCK_STREAM, flags=socket.AI_PASSIVE) DualStackServer.address_family = infos[0][0] HeaderRequestHandler.protocol_version = 'HTTP/1.0' diff --git a/tools/serve_header/serve_header.yml.example b/tools/serve_header/serve_header.yml.example index 42310910e..ec75e2b49 100644 --- a/tools/serve_header/serve_header.yml.example +++ b/tools/serve_header/serve_header.yml.example @@ -10,6 +10,13 @@ # cert_file: localhost.pem # key_file: localhost-key.pem -# address and port for the server to listen on -# bind: null +# address and port for the server to listen on; by default, only this machine +# can connect. Binding to a network address, or to null for all interfaces, +# lets other machines connect, and every request runs make in a working tree. +# bind: localhost # port: 8443 + +# origins whose web pages may read the header (CORS) +# cors_origins: +# - https://godbolt.org +# - https://compiler-explorer.com From 770f62dda9bb06c9772b181f7c6b452ecbe8fd9b Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Thu, 24 Sep 2026 17:02:48 +0200 Subject: [PATCH 26/64] Point to the clang-tidy check that rewrites implicit conversions (#5563) The community-maintained clang-tidy check modernize-nlohmann-json-explicit-conversions rewrites implicit conversions into explicit get() calls, which is exactly the preparation the docs ask for ahead of implicit conversions being switched off by default. Mention it on the JSON_USE_IMPLICIT_CONVERSIONS page and in the migration guide, as promised in discussion #4610. Signed-off-by: Niels Lohmann --- .../docs/api/macros/json_use_implicit_conversions.md | 8 ++++++++ docs/mkdocs/docs/integration/migration_guide.md | 6 ++++++ 2 files changed, 14 insertions(+) diff --git a/docs/mkdocs/docs/api/macros/json_use_implicit_conversions.md b/docs/mkdocs/docs/api/macros/json_use_implicit_conversions.md index 22f6d0072..e3d5fb29d 100644 --- a/docs/mkdocs/docs/api/macros/json_use_implicit_conversions.md +++ b/docs/mkdocs/docs/api/macros/json_use_implicit_conversions.md @@ -24,6 +24,14 @@ By default, implicit conversions are enabled. You can prepare existing code by already defining `JSON_USE_IMPLICIT_CONVERSIONS` to `0` and replace any implicit conversions with calls to [`get`](../basic_json/get.md). +!!! tip "Automatic migration" + + The community-maintained clang-tidy check `modernize-nlohmann-json-explicit-conversions` rewrites implicit + conversions into explicit calls to [`get`](../basic_json/get.md); for example, `#!cpp int i = j;` becomes + `#!cpp int i = j.get();`. The check is not part of clang-tidy itself, and it does not catch every case (for + example, constructing a `std::optional` from a JSON value), so review the result. See + [discussion #4610](https://github.com/nlohmann/json/discussions/4610) for how to build and use it. + !!! hint "CMake option" Implicit conversions can also be controlled with the CMake option diff --git a/docs/mkdocs/docs/integration/migration_guide.md b/docs/mkdocs/docs/integration/migration_guide.md index cb1c62d7b..8b718d969 100644 --- a/docs/mkdocs/docs/integration/migration_guide.md +++ b/docs/mkdocs/docs/integration/migration_guide.md @@ -176,6 +176,12 @@ You can prepare existing code by already defining conversions with calls to [`get`](../api/basic_json/get.md), [`get_to`](../api/basic_json/get_to.md), [`get_ref`](../api/basic_json/get_ref.md), or [`get_ptr`](../api/basic_json/get_ptr.md). +!!! tip "Automatic migration" + + The community-maintained clang-tidy check `modernize-nlohmann-json-explicit-conversions` rewrites most implicit + conversions into calls to [`get`](../api/basic_json/get.md). It is not part of clang-tidy itself; see + [discussion #4610](https://github.com/nlohmann/json/discussions/4610) for how to build and use it. + === "Deprecated" ```cpp From 36c079149a3f9009449d5cca987a404b3eedd99f Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Thu, 24 Sep 2026 17:02:57 +0200 Subject: [PATCH 27/64] Refuse to build the fuzzer drivers with NDEBUG (#5562) The fuzzer drivers check their round trips with assert(), which NDEBUG compiles away. The OSS-Fuzz build keeps assertions on today, but nothing pins that: a build change that adds NDEBUG would silently turn every round-trip check into a mere "does not crash" check. Each driver now stops the build with an #error instead, and includes itself rather than relying on json.hpp. Signed-off-by: Niels Lohmann --- tests/src/fuzzer-parse_bjdata.cpp | 6 ++++++ tests/src/fuzzer-parse_bson.cpp | 6 ++++++ tests/src/fuzzer-parse_cbor.cpp | 6 ++++++ tests/src/fuzzer-parse_json.cpp | 6 ++++++ tests/src/fuzzer-parse_msgpack.cpp | 6 ++++++ tests/src/fuzzer-parse_ubjson.cpp | 6 ++++++ 6 files changed, 36 insertions(+) diff --git a/tests/src/fuzzer-parse_bjdata.cpp b/tests/src/fuzzer-parse_bjdata.cpp index a88479933..41c51a311 100644 --- a/tests/src/fuzzer-parse_bjdata.cpp +++ b/tests/src/fuzzer-parse_bjdata.cpp @@ -46,10 +46,16 @@ The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer drivers. */ +#include #include #include #include +// the round-trip checks below are assertions; NDEBUG would compile them away +#ifdef NDEBUG + #error "the fuzzer drivers must be built without NDEBUG" +#endif + using json = nlohmann::json; // value-stable comparison for the round-trip checks below; see the note diff --git a/tests/src/fuzzer-parse_bson.cpp b/tests/src/fuzzer-parse_bson.cpp index c5f74c7cc..16f36445b 100644 --- a/tests/src/fuzzer-parse_bson.cpp +++ b/tests/src/fuzzer-parse_bson.cpp @@ -19,10 +19,16 @@ The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer drivers. */ +#include #include #include #include +// the round-trip checks below are assertions; NDEBUG would compile them away +#ifdef NDEBUG + #error "the fuzzer drivers must be built without NDEBUG" +#endif + using json = nlohmann::json; // see http://llvm.org/docs/LibFuzzer.html diff --git a/tests/src/fuzzer-parse_cbor.cpp b/tests/src/fuzzer-parse_cbor.cpp index b38e3c1e5..7d599abe2 100644 --- a/tests/src/fuzzer-parse_cbor.cpp +++ b/tests/src/fuzzer-parse_cbor.cpp @@ -19,10 +19,16 @@ The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer drivers. */ +#include #include #include #include +// the round-trip checks below are assertions; NDEBUG would compile them away +#ifdef NDEBUG + #error "the fuzzer drivers must be built without NDEBUG" +#endif + using json = nlohmann::json; // see http://llvm.org/docs/LibFuzzer.html diff --git a/tests/src/fuzzer-parse_json.cpp b/tests/src/fuzzer-parse_json.cpp index 59a278c7b..217d9e0fd 100644 --- a/tests/src/fuzzer-parse_json.cpp +++ b/tests/src/fuzzer-parse_json.cpp @@ -20,10 +20,16 @@ The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer drivers. */ +#include #include #include #include +// the round-trip checks below are assertions; NDEBUG would compile them away +#ifdef NDEBUG + #error "the fuzzer drivers must be built without NDEBUG" +#endif + using json = nlohmann::json; // see http://llvm.org/docs/LibFuzzer.html diff --git a/tests/src/fuzzer-parse_msgpack.cpp b/tests/src/fuzzer-parse_msgpack.cpp index 0b4ab0af7..df961b8d7 100644 --- a/tests/src/fuzzer-parse_msgpack.cpp +++ b/tests/src/fuzzer-parse_msgpack.cpp @@ -19,10 +19,16 @@ The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer drivers. */ +#include #include #include #include +// the round-trip checks below are assertions; NDEBUG would compile them away +#ifdef NDEBUG + #error "the fuzzer drivers must be built without NDEBUG" +#endif + using json = nlohmann::json; // see http://llvm.org/docs/LibFuzzer.html diff --git a/tests/src/fuzzer-parse_ubjson.cpp b/tests/src/fuzzer-parse_ubjson.cpp index 463656c71..20c20eda7 100644 --- a/tests/src/fuzzer-parse_ubjson.cpp +++ b/tests/src/fuzzer-parse_ubjson.cpp @@ -25,10 +25,16 @@ The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer drivers. */ +#include #include #include #include +// the round-trip checks below are assertions; NDEBUG would compile them away +#ifdef NDEBUG + #error "the fuzzer drivers must be built without NDEBUG" +#endif + using json = nlohmann::json; // see http://llvm.org/docs/LibFuzzer.html From 5f659c881a7f08b1664ff7ef600644fd8619034a Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Thu, 24 Sep 2026 17:03:48 +0200 Subject: [PATCH 28/64] Let the labeler assign "aspect: binary formats", "python" and more "CI" (#5557) - "aspect: binary formats" for changes to the binary reader or writer, their tests, fuzzers and docs, or with a binary format in the title; - "python" for Python sources and pip requirements files, matching the label Dependabot sets on its pip updates, so it is never removed there; - "CI" also for changes to the Dependabot and labeler configurations. Signed-off-by: Niels Lohmann --- .github/labeler.yml | 21 +++++++++++++++++++++ 1 file changed, 21 insertions(+) diff --git a/.github/labeler.yml b/.github/labeler.yml index 828660daf..b4c176960 100644 --- a/.github/labeler.yml +++ b/.github/labeler.yml @@ -29,6 +29,27 @@ labels: files: - ".github/external_ci/.*" +- label: "CI" + files: + - ".github/(dependabot|labeler)\\.yml" + +- label: "aspect: binary formats" + files: + - "include/nlohmann/detail/input/binary_reader\\.hpp" + - "include/nlohmann/detail/output/binary_writer\\.hpp" + - "tests/src/unit-(bson|cbor|msgpack|ubjson|bjdata|binary_formats)" + - "tests/src/fuzzer-parse_(bson|cbor|msgpack|ubjson|bjdata)" + - "docs/mkdocs/docs/features/binary_formats/" + - "docs/mkdocs/docs/(api/basic_json|examples)/(to|from)_(bson|cbor|msgpack|ubjson|bjdata)" + +- label: "aspect: binary formats" + title: "(?i)(bson|cbor|msgpack|messagepack|ubjson|bjdata|binary format)" + +- label: "python" + files: + - "\\.py$" + - "requirements[^/]*\\.txt$" + - label: "S" size-below: 10 - label: "M" From 8699de30642500fb8c940aee9c9812bb3f5af857 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Thu, 24 Sep 2026 17:04:44 +0200 Subject: [PATCH 29/64] Stop allocating the BJData excluded-marker list per container (#5555) write_ubjson() built a std::vector of the eight markers BJData forbids as the type of an optimized container - one heap allocation plus a linear search for every array and object it wrote with use_type, even for plain UBJSON output, where the list isn't consulted. The list was also spelled out twice. A constexpr helper, is_bjdata_excluded_type_marker(), replaces both. Signed-off-by: Niels Lohmann --- .../nlohmann/detail/output/binary_writer.hpp | 23 ++++++++++++++----- single_include/nlohmann/json.hpp | 23 ++++++++++++++----- 2 files changed, 34 insertions(+), 12 deletions(-) diff --git a/include/nlohmann/detail/output/binary_writer.hpp b/include/nlohmann/detail/output/binary_writer.hpp index 495f872c0..eaa8d63b6 100644 --- a/include/nlohmann/detail/output/binary_writer.hpp +++ b/include/nlohmann/detail/output/binary_writer.hpp @@ -881,8 +881,6 @@ class binary_writer return ubjson_prefix(v, use_bjdata) == first_prefix; }); - std::vector bjdx = {'[', '{', 'S', 'H', 'T', 'F', 'N', 'Z'}; // excluded markers in bjdata optimized type - // an optimized array of a valueless type carries no payload, so a // reader has nothing but the declared count to bound the allocation // by and refuses an excessive one. Write the unoptimized form for @@ -893,7 +891,7 @@ class binary_writer && j.m_data.m_value.array->size() > detail::max_valueless_container_size; if (same_prefix && !excessive_valueless - && !(use_bjdata && std::find(bjdx.begin(), bjdx.end(), first_prefix) != bjdx.end())) + && !(use_bjdata && is_bjdata_excluded_type_marker(first_prefix))) { prefix_required = false; oa.write_character(to_char_type('$')); @@ -997,9 +995,7 @@ class binary_writer return ubjson_prefix(v, use_bjdata) == first_prefix; }); - std::vector bjdx = {'[', '{', 'S', 'H', 'T', 'F', 'N', 'Z'}; // excluded markers in bjdata optimized type - - if (same_prefix && !(use_bjdata && std::find(bjdx.begin(), bjdx.end(), first_prefix) != bjdx.end())) + if (same_prefix && !(use_bjdata && is_bjdata_excluded_type_marker(first_prefix))) { prefix_required = false; oa.write_character(to_char_type('$')); @@ -1726,6 +1722,21 @@ class binary_writer } } + /*! + @brief whether BJData forbids @a marker as the type of an optimized array + or object + + Containers, strings, high-precision numbers, booleans and null cannot be + declared as the single type of an optimized container in BJData; such a + container is written unoptimized. The reader rejects them with the same + list (binary_reader::bjd_optimized_type_markers). + */ + static constexpr bool is_bjdata_excluded_type_marker(const CharType marker) noexcept + { + return marker == '[' || marker == '{' || marker == 'S' || marker == 'H' + || marker == 'T' || marker == 'F' || marker == 'N' || marker == 'Z'; + } + static constexpr CharType get_ubjson_float_prefix(float /*unused*/) { return 'd'; // float 32 diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 1d5e21927..aa2916aa5 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -19745,8 +19745,6 @@ class binary_writer return ubjson_prefix(v, use_bjdata) == first_prefix; }); - std::vector bjdx = {'[', '{', 'S', 'H', 'T', 'F', 'N', 'Z'}; // excluded markers in bjdata optimized type - // an optimized array of a valueless type carries no payload, so a // reader has nothing but the declared count to bound the allocation // by and refuses an excessive one. Write the unoptimized form for @@ -19757,7 +19755,7 @@ class binary_writer && j.m_data.m_value.array->size() > detail::max_valueless_container_size; if (same_prefix && !excessive_valueless - && !(use_bjdata && std::find(bjdx.begin(), bjdx.end(), first_prefix) != bjdx.end())) + && !(use_bjdata && is_bjdata_excluded_type_marker(first_prefix))) { prefix_required = false; oa.write_character(to_char_type('$')); @@ -19861,9 +19859,7 @@ class binary_writer return ubjson_prefix(v, use_bjdata) == first_prefix; }); - std::vector bjdx = {'[', '{', 'S', 'H', 'T', 'F', 'N', 'Z'}; // excluded markers in bjdata optimized type - - if (same_prefix && !(use_bjdata && std::find(bjdx.begin(), bjdx.end(), first_prefix) != bjdx.end())) + if (same_prefix && !(use_bjdata && is_bjdata_excluded_type_marker(first_prefix))) { prefix_required = false; oa.write_character(to_char_type('$')); @@ -20590,6 +20586,21 @@ class binary_writer } } + /*! + @brief whether BJData forbids @a marker as the type of an optimized array + or object + + Containers, strings, high-precision numbers, booleans and null cannot be + declared as the single type of an optimized container in BJData; such a + container is written unoptimized. The reader rejects them with the same + list (binary_reader::bjd_optimized_type_markers). + */ + static constexpr bool is_bjdata_excluded_type_marker(const CharType marker) noexcept + { + return marker == '[' || marker == '{' || marker == 'S' || marker == 'H' + || marker == 'T' || marker == 'F' || marker == 'N' || marker == 'Z'; + } + static constexpr CharType get_ubjson_float_prefix(float /*unused*/) { return 'd'; // float 32 From 2e91641de27335ff48bca48c2b2281f2688b1e99 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Thu, 24 Sep 2026 17:05:37 +0200 Subject: [PATCH 30/64] Test JSON_BRACE_INIT_COPY_SEMANTICS for real, and fix one-element tuples under it (#5544) * Test JSON_BRACE_INIT_COPY_SEMANTICS for real, and fix one-element tuples under it The opt-in JSON_BRACE_INIT_COPY_SEMANTICS was never exercised by CI: - Its only test, in unit-regression3.cpp, was guarded by `#if defined(JSON_BRACE_INIT_COPY_SEMANTICS)` after the #include. The header #undefs the macro unconditionally in macro_unscope.hpp, so the guard was always false and the test compiled to nothing, whatever -D flag was passed. - The ci_test_brace_init_copy_semantics target that passes the flag was not named by any workflow. Move the test into its own translation unit that defines the macro before including the header, as unit-diagnostics.cpp does for JSON_DIAGNOSTICS. It now runs in every CI job and for every standard. Remove the unused target: it ran the whole suite with the macro, and that suite deliberately relies on default brace-init semantics in about 90 places (e.g. `json({1})` meaning `[1]`), so it could never pass. Running the whole suite with the macro did find one library bug: to_json for std::tuple builds `j = { std::get(t)... }`, so with copy semantics a one-element tuple became its element. `json(std::tuple{5})` was `5` instead of `[5]`, and `get>()` threw type_error.302 on the result. Under the macro, a one-element tuple now builds exactly what the default deduction builds. Without the macro nothing changes. The new tests also pin that the library's other conversions produce the same values with and without the macro. The macro page now says that the macro affects every single-element list (`json j = {1}` is `1`), and that all translation units must agree on it, since it has no ABI tag. Signed-off-by: Niels Lohmann * Make JSON_BRACE_INIT_COPY_SEMANTICS part of the ABI tag The macro changes the body of the initializer-list constructor and adds a to_json_tuple_impl overload, both with the same mangled names in either mode, so mixing translation units silently picked one definition. Encode it in the inline namespace as `_bics`, as JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON does with `_ldvcmp`. The macro is new in the unreleased 3.13.0, so no existing namespace name changes. - Move the macro's default into abi_macros.hpp so json_fwd.hpp computes the same namespace, and keep it defined under JSON_TEST_KEEP_MACROS. - Check the tag in the ABI config tests and in the unit test. - List `_bics` (and the missing `_dp`) in the namespace docs and in the natvis generator; regenerate nlohmann_json.natvis. - Replace the "define it consistently" warning with an ABI note. Suggested by @gregmarr in the review of #5544. Signed-off-by: Niels Lohmann * Fix the cppcheck, clang-tidy and legacy-comparison CI failures - to_json_tuple_impl() moved the element in both branches of a ternary; only one runs, but cppcheck reported accessMoved. Use if/else. - The ABI tag test looked for "json_abi_bics", which misses when another tag comes first, as in json_abi_ldvcmp_bics; look for "_bics". - readability-qualified-auto in the items() test. Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- cmake/ci.cmake | 15 - .../macros/json_brace_init_copy_semantics.md | 22 + docs/mkdocs/docs/features/namespace.md | 2 + include/nlohmann/detail/abi_macros.hpp | 19 +- .../nlohmann/detail/conversions/to_json.hpp | 24 + include/nlohmann/detail/macro_scope.hpp | 4 - include/nlohmann/detail/macro_unscope.hpp | 2 +- nlohmann_json.natvis | 720 ++++++++++++++++++ single_include/nlohmann/json.hpp | 49 +- single_include/nlohmann/json_fwd.hpp | 19 +- tests/abi/config/default.cpp | 4 + tests/abi/config/noversion.cpp | 4 + tests/src/unit-brace-init-copy-semantics.cpp | 167 ++++ tests/src/unit-regression3.cpp | 27 - tools/generate_natvis/generate_natvis.py | 2 +- 15 files changed, 1015 insertions(+), 65 deletions(-) create mode 100644 tests/src/unit-brace-init-copy-semantics.cpp diff --git a/cmake/ci.cmake b/cmake/ci.cmake index f854138b6..a99788633 100644 --- a/cmake/ci.cmake +++ b/cmake/ci.cmake @@ -230,21 +230,6 @@ add_custom_target(ci_test_simdutf COMMENT "Compile and test with simdutf UTF-8 validation enabled" ) -############################################################################### -# Enable brace-init copy semantics. -############################################################################### - -add_custom_target(ci_test_brace_init_copy_semantics - COMMAND ${CMAKE_COMMAND} - -DCMAKE_BUILD_TYPE=Debug -GNinja - -DJSON_BuildTests=ON -DJSON_FastTests=ON - -DCMAKE_CXX_FLAGS=-DJSON_BRACE_INIT_COPY_SEMANTICS=1 - -S${PROJECT_SOURCE_DIR} -B${PROJECT_BINARY_DIR}/build_brace_init_copy_semantics - COMMAND ${CMAKE_COMMAND} --build ${PROJECT_BINARY_DIR}/build_brace_init_copy_semantics - COMMAND cd ${PROJECT_BINARY_DIR}/build_brace_init_copy_semantics && ${CMAKE_CTEST_COMMAND} --parallel ${N} --output-on-failure - COMMENT "Compile and test with brace-init copy semantics enabled" -) - ############################################################################### # Enable strict NUL-byte handling. ############################################################################### diff --git a/docs/mkdocs/docs/api/macros/json_brace_init_copy_semantics.md b/docs/mkdocs/docs/api/macros/json_brace_init_copy_semantics.md index 970c20537..2301a0486 100644 --- a/docs/mkdocs/docs/api/macros/json_brace_init_copy_semantics.md +++ b/docs/mkdocs/docs/api/macros/json_brace_init_copy_semantics.md @@ -38,6 +38,28 @@ The default value is `0` (disabled — existing behavior is preserved). This macro must be defined **before** including ``. Defining it after the include has no effect. +!!! warning "Applies to every single-element list" + + The macro does not only affect a single JSON value in braces. **Any** single-element braced list is treated as its + element, so it no longer creates a one-element array: + + ```cpp + json j1 = {1}; // 1, not [1] + json j2 = {"text"}; // "text", not ["text"] + json j3 = {{1, 2}}; // [1,2], not [[1,2]] + ``` + + Code that relies on these producing arrays must use `json::array()` instead (see below). Lists with more than one + element, and a single `[string, value]` pair such as `{{"key", "value"}}`, which still creates an object, are not + affected. The library's own conversions are not affected either: for example, `std::tuple{5}` still becomes + `[5]`. + +!!! note "ABI compatibility" + + The value of this macro is encoded in the [namespace](../../features/namespace.md) (tag `_bics`), resulting in + distinct symbol names. Translation units compiled with and without it can therefore be linked into the same program + without One Definition Rule (ODR) violations, but they cannot exchange instances of library types. + !!! tip "Workaround without the macro" To explicitly create a single-element array without enabling this macro, use `json::array()`: diff --git a/docs/mkdocs/docs/features/namespace.md b/docs/mkdocs/docs/features/namespace.md index c4efe772a..09e53f3a2 100644 --- a/docs/mkdocs/docs/features/namespace.md +++ b/docs/mkdocs/docs/features/namespace.md @@ -16,6 +16,8 @@ The complete default namespace name is derived as follows: - [`JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON`](../api/macros/json_use_legacy_discarded_value_comparison.md) defined non-zero appends `_ldvcmp`. - [`JSON_DIAGNOSTIC_POSITIONS`](../api/macros/json_diagnostic_positions.md) defined non-zero appends `_dp`. + - [`JSON_BRACE_INIT_COPY_SEMANTICS`](../api/macros/json_brace_init_copy_semantics.md) defined non-zero appends + `_bics`. - The inline namespace ends with the suffix `_v` followed by the 3 components of the version number separated by underscores. To omit the version component, see [Disabling the version component](#disabling-the-version-component) below. diff --git a/include/nlohmann/detail/abi_macros.hpp b/include/nlohmann/detail/abi_macros.hpp index 3e07a6a98..cca04e8ec 100644 --- a/include/nlohmann/detail/abi_macros.hpp +++ b/include/nlohmann/detail/abi_macros.hpp @@ -34,6 +34,10 @@ #define JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON 0 #endif +#ifndef JSON_BRACE_INIT_COPY_SEMANTICS + #define JSON_BRACE_INIT_COPY_SEMANTICS 0 +#endif + #if JSON_DIAGNOSTICS #define NLOHMANN_JSON_ABI_TAG_DIAGNOSTICS _diag #else @@ -52,20 +56,27 @@ #define NLOHMANN_JSON_ABI_TAG_LEGACY_DISCARDED_VALUE_COMPARISON #endif +#if JSON_BRACE_INIT_COPY_SEMANTICS + #define NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS _bics +#else + #define NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS +#endif + #ifndef NLOHMANN_JSON_NAMESPACE_NO_VERSION #define NLOHMANN_JSON_NAMESPACE_NO_VERSION 0 #endif // Construct the namespace ABI tags component -#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c) json_abi ## a ## b ## c -#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c) \ - NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c) +#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d) json_abi ## a ## b ## c ## d +#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d) \ + NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d) #define NLOHMANN_JSON_ABI_TAGS \ NLOHMANN_JSON_ABI_TAGS_CONCAT( \ NLOHMANN_JSON_ABI_TAG_DIAGNOSTICS, \ NLOHMANN_JSON_ABI_TAG_LEGACY_DISCARDED_VALUE_COMPARISON, \ - NLOHMANN_JSON_ABI_TAG_DIAGNOSTIC_POSITIONS) + NLOHMANN_JSON_ABI_TAG_DIAGNOSTIC_POSITIONS, \ + NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS) // Construct the namespace version component #define NLOHMANN_JSON_NAMESPACE_VERSION_CONCAT_EX(major, minor, patch) \ diff --git a/include/nlohmann/detail/conversions/to_json.hpp b/include/nlohmann/detail/conversions/to_json.hpp index 5f8644700..491bb9873 100644 --- a/include/nlohmann/detail/conversions/to_json.hpp +++ b/include/nlohmann/detail/conversions/to_json.hpp @@ -471,6 +471,30 @@ inline void to_json_tuple_impl(BasicJsonType& j, const Tuple& t, index_sequence< j = { std::get(t)... }; } +#if JSON_BRACE_INIT_COPY_SEMANTICS +// JSON_BRACE_INIT_COPY_SEMANTICS makes a one-element braced list copy its +// element instead of wrapping it, which would serialize std::tuple{5} as 5 +// rather than [5]. Build what the default deduction builds instead: an object +// if the element is a [string, value] pair, a one-element array otherwise. +template +inline void to_json_tuple_impl(BasicJsonType& j, const Tuple& t, index_sequence<0> /*unused*/) +{ + BasicJsonType element(std::get<0>(t)); + // same test as the initializer-list constructor, including the cast that + // keeps a string type constructible from 0 from selecting operator[](key) + const bool is_member = element.is_array() && element.size() == 2 + && element[static_cast(0)].is_string(); + if (is_member) + { + j = BasicJsonType::object({std::move(element)}); + } + else + { + j = BasicJsonType::array({std::move(element)}); + } +} +#endif + template inline void to_json_tuple_impl(BasicJsonType& j, const Tuple& /*unused*/, index_sequence<> /*unused*/) { diff --git a/include/nlohmann/detail/macro_scope.hpp b/include/nlohmann/detail/macro_scope.hpp index def9da6f8..96fa165f5 100644 --- a/include/nlohmann/detail/macro_scope.hpp +++ b/include/nlohmann/detail/macro_scope.hpp @@ -813,10 +813,6 @@ void templated_json_throw(ExceptionType exception) #define JSON_USE_GLOBAL_UDLS 1 #endif -#ifndef JSON_BRACE_INIT_COPY_SEMANTICS - #define JSON_BRACE_INIT_COPY_SEMANTICS 0 -#endif - #ifndef JSON_STRICT_NUL_HANDLING #define JSON_STRICT_NUL_HANDLING 0 #endif diff --git a/include/nlohmann/detail/macro_unscope.hpp b/include/nlohmann/detail/macro_unscope.hpp index c692ea68e..afcbfc38b 100644 --- a/include/nlohmann/detail/macro_unscope.hpp +++ b/include/nlohmann/detail/macro_unscope.hpp @@ -26,7 +26,6 @@ #undef JSON_NO_UNIQUE_ADDRESS #undef JSON_DISABLE_ENUM_SERIALIZATION #undef JSON_USE_GLOBAL_UDLS -#undef JSON_BRACE_INIT_COPY_SEMANTICS #undef JSON_STRICT_NUL_HANDLING #ifndef JSON_TEST_KEEP_MACROS @@ -45,6 +44,7 @@ #undef JSON_HAS_STD_FORMAT #undef JSON_HAS_STATIC_RTTI #undef JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON + #undef JSON_BRACE_INIT_COPY_SEMANTICS #endif #include diff --git a/nlohmann_json.natvis b/nlohmann_json.natvis index 09a46d67d..2eccbe17c 100644 --- a/nlohmann_json.natvis +++ b/nlohmann_json.natvis @@ -215,6 +215,126 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -275,4 +395,604 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index aa2916aa5..c11fa27e5 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -91,6 +91,10 @@ #define JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON 0 #endif +#ifndef JSON_BRACE_INIT_COPY_SEMANTICS + #define JSON_BRACE_INIT_COPY_SEMANTICS 0 +#endif + #if JSON_DIAGNOSTICS #define NLOHMANN_JSON_ABI_TAG_DIAGNOSTICS _diag #else @@ -109,20 +113,27 @@ #define NLOHMANN_JSON_ABI_TAG_LEGACY_DISCARDED_VALUE_COMPARISON #endif +#if JSON_BRACE_INIT_COPY_SEMANTICS + #define NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS _bics +#else + #define NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS +#endif + #ifndef NLOHMANN_JSON_NAMESPACE_NO_VERSION #define NLOHMANN_JSON_NAMESPACE_NO_VERSION 0 #endif // Construct the namespace ABI tags component -#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c) json_abi ## a ## b ## c -#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c) \ - NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c) +#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d) json_abi ## a ## b ## c ## d +#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d) \ + NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d) #define NLOHMANN_JSON_ABI_TAGS \ NLOHMANN_JSON_ABI_TAGS_CONCAT( \ NLOHMANN_JSON_ABI_TAG_DIAGNOSTICS, \ NLOHMANN_JSON_ABI_TAG_LEGACY_DISCARDED_VALUE_COMPARISON, \ - NLOHMANN_JSON_ABI_TAG_DIAGNOSTIC_POSITIONS) + NLOHMANN_JSON_ABI_TAG_DIAGNOSTIC_POSITIONS, \ + NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS) // Construct the namespace version component #define NLOHMANN_JSON_NAMESPACE_VERSION_CONCAT_EX(major, minor, patch) \ @@ -3191,10 +3202,6 @@ void templated_json_throw(ExceptionType exception) #define JSON_USE_GLOBAL_UDLS 1 #endif -#ifndef JSON_BRACE_INIT_COPY_SEMANTICS - #define JSON_BRACE_INIT_COPY_SEMANTICS 0 -#endif - #ifndef JSON_STRICT_NUL_HANDLING #define JSON_STRICT_NUL_HANDLING 0 #endif @@ -6767,6 +6774,30 @@ inline void to_json_tuple_impl(BasicJsonType& j, const Tuple& t, index_sequence< j = { std::get(t)... }; } +#if JSON_BRACE_INIT_COPY_SEMANTICS +// JSON_BRACE_INIT_COPY_SEMANTICS makes a one-element braced list copy its +// element instead of wrapping it, which would serialize std::tuple{5} as 5 +// rather than [5]. Build what the default deduction builds instead: an object +// if the element is a [string, value] pair, a one-element array otherwise. +template +inline void to_json_tuple_impl(BasicJsonType& j, const Tuple& t, index_sequence<0> /*unused*/) +{ + BasicJsonType element(std::get<0>(t)); + // same test as the initializer-list constructor, including the cast that + // keeps a string type constructible from 0 from selecting operator[](key) + const bool is_member = element.is_array() && element.size() == 2 + && element[static_cast(0)].is_string(); + if (is_member) + { + j = BasicJsonType::object({std::move(element)}); + } + else + { + j = BasicJsonType::array({std::move(element)}); + } +} +#endif + template inline void to_json_tuple_impl(BasicJsonType& j, const Tuple& /*unused*/, index_sequence<> /*unused*/) { @@ -30395,7 +30426,6 @@ struct formatter // NOLINT(cert-dcl58-c #undef JSON_NO_UNIQUE_ADDRESS #undef JSON_DISABLE_ENUM_SERIALIZATION #undef JSON_USE_GLOBAL_UDLS -#undef JSON_BRACE_INIT_COPY_SEMANTICS #undef JSON_STRICT_NUL_HANDLING #ifndef JSON_TEST_KEEP_MACROS @@ -30414,6 +30444,7 @@ struct formatter // NOLINT(cert-dcl58-c #undef JSON_HAS_STD_FORMAT #undef JSON_HAS_STATIC_RTTI #undef JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON + #undef JSON_BRACE_INIT_COPY_SEMANTICS #endif // #include diff --git a/single_include/nlohmann/json_fwd.hpp b/single_include/nlohmann/json_fwd.hpp index 525e65b64..281c05efa 100644 --- a/single_include/nlohmann/json_fwd.hpp +++ b/single_include/nlohmann/json_fwd.hpp @@ -52,6 +52,10 @@ #define JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON 0 #endif +#ifndef JSON_BRACE_INIT_COPY_SEMANTICS + #define JSON_BRACE_INIT_COPY_SEMANTICS 0 +#endif + #if JSON_DIAGNOSTICS #define NLOHMANN_JSON_ABI_TAG_DIAGNOSTICS _diag #else @@ -70,20 +74,27 @@ #define NLOHMANN_JSON_ABI_TAG_LEGACY_DISCARDED_VALUE_COMPARISON #endif +#if JSON_BRACE_INIT_COPY_SEMANTICS + #define NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS _bics +#else + #define NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS +#endif + #ifndef NLOHMANN_JSON_NAMESPACE_NO_VERSION #define NLOHMANN_JSON_NAMESPACE_NO_VERSION 0 #endif // Construct the namespace ABI tags component -#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c) json_abi ## a ## b ## c -#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c) \ - NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c) +#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d) json_abi ## a ## b ## c ## d +#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d) \ + NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d) #define NLOHMANN_JSON_ABI_TAGS \ NLOHMANN_JSON_ABI_TAGS_CONCAT( \ NLOHMANN_JSON_ABI_TAG_DIAGNOSTICS, \ NLOHMANN_JSON_ABI_TAG_LEGACY_DISCARDED_VALUE_COMPARISON, \ - NLOHMANN_JSON_ABI_TAG_DIAGNOSTIC_POSITIONS) + NLOHMANN_JSON_ABI_TAG_DIAGNOSTIC_POSITIONS, \ + NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS) // Construct the namespace version component #define NLOHMANN_JSON_NAMESPACE_VERSION_CONCAT_EX(major, minor, patch) \ diff --git a/tests/abi/config/default.cpp b/tests/abi/config/default.cpp index f3ee23110..d0b4ba54b 100644 --- a/tests/abi/config/default.cpp +++ b/tests/abi/config/default.cpp @@ -32,6 +32,10 @@ TEST_CASE("default namespace") expected += "_dp"; #endif +#if JSON_BRACE_INIT_COPY_SEMANTICS + expected += "_bics"; +#endif + expected += "_v" STRINGIZE(NLOHMANN_JSON_VERSION_MAJOR); expected += "_" STRINGIZE(NLOHMANN_JSON_VERSION_MINOR); expected += "_" STRINGIZE(NLOHMANN_JSON_VERSION_PATCH) "::basic_json"; diff --git a/tests/abi/config/noversion.cpp b/tests/abi/config/noversion.cpp index cbdcb149b..789107181 100644 --- a/tests/abi/config/noversion.cpp +++ b/tests/abi/config/noversion.cpp @@ -33,6 +33,10 @@ TEST_CASE("default namespace without version component") expected += "_dp"; #endif +#if JSON_BRACE_INIT_COPY_SEMANTICS + expected += "_bics"; +#endif + expected += "::basic_json"; // fallback for Clang diff --git a/tests/src/unit-brace-init-copy-semantics.cpp b/tests/src/unit-brace-init-copy-semantics.cpp new file mode 100644 index 000000000..1ee0c6607 --- /dev/null +++ b/tests/src/unit-brace-init-copy-semantics.cpp @@ -0,0 +1,167 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#include "doctest_compatibility.h" + +// This file tests the opt-in JSON_BRACE_INIT_COPY_SEMANTICS, so it defines the +// macro itself rather than relying on a -D flag, and runs in every build. +#ifdef JSON_BRACE_INIT_COPY_SEMANTICS + #undef JSON_BRACE_INIT_COPY_SEMANTICS +#endif + +#define JSON_BRACE_INIT_COPY_SEMANTICS 1 + +#include +using nlohmann::json; + +#include +#include +#include +#include +#include +#include +#include + +#define STRINGIZE_EX(x) #x +#define STRINGIZE(x) STRINGIZE_EX(x) + +TEST_CASE("JSON_BRACE_INIT_COPY_SEMANTICS") +{ + SECTION("the macro is part of the ABI tag") + { + const std::string ns = STRINGIZE(NLOHMANN_JSON_NAMESPACE); + // other tags may come before it, e.g. json_abi_ldvcmp_bics + CHECK(ns.find("_bics") != std::string::npos); + } + + SECTION("single-element brace initialization copies the element (#5074)") + { + json const j_obj = {{"key", "value"}, {"num", 42}}; + json const j_arr = {1, 2, 3}; + + // object: brace init copies instead of wrapping + json const j1{j_obj}; + CHECK(j1.is_object()); + CHECK(j1 == j_obj); + + // array: brace init copies instead of wrapping + json const j2{j_arr}; + CHECK(j2.is_array()); + CHECK(j2.size() == 3); + CHECK(j2 == j_arr); + + // this applies to any single element, not only to JSON values + json const j3{true}; + CHECK(j3.is_boolean()); + + json const j4{42}; + CHECK(j4.is_number_integer()); + + json const j5 = {1}; + CHECK(j5 == 1); + + json const j6 = {"text"}; + CHECK(j6 == "text"); + + json const j7 = {{1, 2}}; + CHECK(j7 == json::array({1, 2})); + } + + SECTION("what the macro does not change") + { + // lists with more than one element are unaffected + json const j1 = {1, 2}; + CHECK(j1.is_array()); + CHECK(j1.size() == 2); + + // a single [string, value] pair still describes an object + json const j2 = {{"key", "value"}}; + CHECK(j2.is_object()); + CHECK(j2["key"] == "value"); + + // json::array() always creates an array + json const j3 = json::array({1}); + CHECK(j3.is_array()); + CHECK(j3.size() == 1); + CHECK(j3[0] == 1); + + json const j_obj = {{"key", "value"}}; + json const j4 = json::array({j_obj}); + CHECK(j4.is_array()); + CHECK(j4.size() == 1); + CHECK(j4[0] == j_obj); + } + + SECTION("conversions build the same values as without the macro") + { + SECTION("one-element std::tuple") + { + json const j1 = std::tuple {5}; + CHECK(j1.dump() == "[5]"); + CHECK(std::get<0>(j1.get>()) == 5); + + json const j2 = std::tuple {"text"}; + CHECK(j2.dump() == "[\"text\"]"); + CHECK(std::get<0>(j2.get>()) == "text"); + + json const j3 = std::tuple {json::array({1, 2})}; + CHECK(j3.dump() == "[[1,2]]"); + + // as without the macro, a [string, value] pair becomes an object + // member (see the known limitation documented for std::pair) + json const j4 = std::tuple> {{"a", 1}}; + CHECK(j4.dump() == "{\"a\":1}"); + } + + SECTION("tuples with more elements") + { + json const j1 = std::tuple {1, "a"}; + CHECK(j1.dump() == "[1,\"a\"]"); + + json const j2 = std::tuple<> {}; + CHECK(j2.dump() == "[]"); + } + + SECTION("one-element containers") + { + json const j1 = std::vector {1}; + CHECK(j1.dump() == "[1]"); + CHECK(j1.get>() == std::vector {1}); + + std::array const arr = {{1}}; + json const j2 = arr; + CHECK(j2.dump() == "[1]"); + + json const j3 = std::list {"a"}; + CHECK(j3.dump() == "[\"a\"]"); + + json const j4 = std::map {{"a", 1}}; + CHECK(j4.dump() == "{\"a\":1}"); + + json const j5 = std::map {{1, 2}}; + CHECK(j5.dump() == "[[1,2]]"); + } + + SECTION("std::pair") + { + json const j = std::pair {1, 2}; + CHECK(j.dump() == "[1,2]"); + CHECK((j.get>() == std::pair {1, 2})); + } + + SECTION("items()") + { + json j_obj = {{"key", 1}}; + for (const auto& el : j_obj.items()) + { + json const j = el; + CHECK(j.dump() == "{\"key\":1}"); + } + } + } +} diff --git a/tests/src/unit-regression3.cpp b/tests/src/unit-regression3.cpp index a5be9ec4b..882b3866b 100644 --- a/tests/src/unit-regression3.cpp +++ b/tests/src/unit-regression3.cpp @@ -658,33 +658,6 @@ TEST_CASE("regression test #5074 - portable workaround for single-element brace CHECK(j[0] == j_obj); } -#if defined(JSON_BRACE_INIT_COPY_SEMANTICS) && (JSON_BRACE_INIT_COPY_SEMANTICS == 1) -TEST_CASE("regression test #5074 - single-element brace init with JSON_BRACE_INIT_COPY_SEMANTICS") -{ - // with JSON_BRACE_INIT_COPY_SEMANTICS: single-element brace init copies/moves - json const j_obj = {{"key", "value"}, {"num", 42}}; - json const j_arr = {1, 2, 3}; - - // object: brace init copies instead of wrapping - json const j1{j_obj}; - CHECK(j1.is_object()); - CHECK(j1 == j_obj); - - // array: brace init copies instead of wrapping - json const j2{j_arr}; - CHECK(j2.is_array()); - CHECK(j2.size() == 3); - CHECK(j2 == j_arr); - - // primitives still work as initializer lists - json const j3{true}; - CHECK(j3.is_boolean()); - - json const j4{42}; - CHECK(j4.is_number_integer()); -} -#endif - struct Example_5122 { float b = 2; diff --git a/tools/generate_natvis/generate_natvis.py b/tools/generate_natvis/generate_natvis.py index 9266050c5..968690abe 100755 --- a/tools/generate_natvis/generate_natvis.py +++ b/tools/generate_natvis/generate_natvis.py @@ -20,7 +20,7 @@ if __name__ == '__main__': namespaces = ['nlohmann'] abi_prefix = 'json_abi' - abi_tags = ['_diag', '_ldvcmp'] + abi_tags = ['_diag', '_ldvcmp', '_dp', '_bics'] version = '_v' + args.version.replace('.', '_') inline_namespaces = [] From 305ca7dadd17bbbf215822e8e8e9c792c182cdaa Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Thu, 24 Sep 2026 17:07:21 +0200 Subject: [PATCH 31/64] Add missing headers to BUILD.bazel and check it in CI (#5554) * Add missing headers to BUILD.bazel and make its generator reproduce it The "json" cc_library did not list three headers that the library includes: - detail/meta/logic.hpp (added in #5016, included by from_json.hpp) - detail/input/number_parse.hpp (added in #5283, included by lexer.hpp) - detail/input/string_scan.hpp (added in #5283, included by lexer.hpp and serializer.hpp) Bazel's sandbox only exposes declared headers, so any target depending on @nlohmann_json//:json and including failed with "'nlohmann/detail/meta/logic.hpp' file not found". The file could not simply be regenerated, because the generator behind "make BUILD.bazel" was stale: it wrote only the "json" cc_library and dropped the load() statements, the license block, and the "singleheader-json" target that were added by hand in #4584. The generator now emits the complete file, so its output differs from the previous BUILD.bazel only by the three headers. It also resolves the glob against the project root instead of the working directory and sorts the list explicitly. "make BUILD.bazel" is now phony: in a fresh checkout, BUILD.bazel is not older than the headers, so make considered it up to date, and a removed header would never trigger a rebuild. "make check-amalgamation" also checks that BUILD.bazel is up to date. Signed-off-by: Niels Lohmann * Check in CI that BUILD.bazel is up to date The "Check amalgamation" workflow now also regenerates BUILD.bazel, so a pull request that adds, renames, or removes a header without updating the Bazel header list fails, and the attached amalgamation.patch contains the fix. The failure comment and the contribution guidelines mention the new check, and the comment now links to the existing "Amalgamate the source code" section instead of the "Files to change" anchor that was removed in #4560. Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- .github/CONTRIBUTING.md | 9 ++++ .github/workflows/check_amalgamation.yml | 7 ++- .../workflows/comment_check_amalgamation.yml | 4 +- BUILD.bazel | 3 ++ FILES.md | 6 ++- Makefile | 12 +++-- cmake/scripts/gen_bazel_build_file.cmake | 44 ++++++++++++++++--- 7 files changed, 72 insertions(+), 13 deletions(-) diff --git a/.github/CONTRIBUTING.md b/.github/CONTRIBUTING.md index 68f82c474..4125b066a 100644 --- a/.github/CONTRIBUTING.md +++ b/.github/CONTRIBUTING.md @@ -158,6 +158,15 @@ make amalgamate Running `make amalgamate` will also apply automatic formatting to the source files using [`Artistic Style`](https://astyle.sourceforge.net/). This formatting may modify your source files in-place. Be certain to review and commit any changes to avoid unintended formatting diffs in commits. +If you add, rename, or remove a header in `include/nlohmann`, also regenerate the header list in +[`BUILD.bazel`](https://github.com/nlohmann/json/blob/develop/BUILD.bazel) (requires CMake) by executing: + +```shell +make BUILD.bazel +``` + +The amalgamation check in CI fails if any of these generated files is out of date. + ## Recommended documentation - The library’s [README file](https://github.com/nlohmann/json/blob/master/README.md) is an excellent starting point to diff --git a/.github/workflows/check_amalgamation.yml b/.github/workflows/check_amalgamation.yml index f70ebfba0..f692e434a 100644 --- a/.github/workflows/check_amalgamation.yml +++ b/.github/workflows/check_amalgamation.yml @@ -57,13 +57,16 @@ jobs: python3 -mvenv venv venv/bin/pip3 install -r $MAIN_DIR/tools/astyle/requirements.txt - - name: Regenerate amalgamation and formatting + - name: Regenerate amalgamation, formatting, and BUILD.bazel run: | cd $MAIN_DIR python3 $TOOL_DIR/amalgamate.py -c $TOOL_DIR/config_json.json -s . python3 $TOOL_DIR/amalgamate.py -c $TOOL_DIR/config_json_fwd.json -s . + # the header list of the Bazel "json" target must match the files in include/ + cmake -P cmake/scripts/gen_bazel_build_file.cmake + ${{ github.workspace }}/venv/bin/astyle --project=tools/astyle/.astylerc --suffix=none --quiet \ $INCLUDE_DIR/json.hpp $INCLUDE_DIR/json_fwd.hpp @@ -87,7 +90,7 @@ jobs: mkdir -p ${{ github.workspace }}/patch git diff --patch --no-color > ${{ github.workspace }}/patch/amalgamation.patch if [ -s ${{ github.workspace }}/patch/amalgamation.patch ]; then - echo "The source code has not been amalgamated/formatted correctly. Diff:" + echo "The source code has not been amalgamated/formatted correctly or BUILD.bazel is out of date. Diff:" cat ${{ github.workspace }}/patch/amalgamation.patch echo "has_diff=true" >> "$GITHUB_OUTPUT" else diff --git a/.github/workflows/comment_check_amalgamation.yml b/.github/workflows/comment_check_amalgamation.yml index 788c1b8ce..4667329d2 100644 --- a/.github/workflows/comment_check_amalgamation.yml +++ b/.github/workflows/comment_check_amalgamation.yml @@ -95,13 +95,13 @@ jobs: issue_number: issue_number, owner: context.repo.owner, repo: context.repo.repo, - body: '## 🔴 Amalgamation check failed! 🔴\nThe source code has not been amalgamated and/or formatted correctly.' + body: '## 🔴 Amalgamation check failed! 🔴\nThe source code has not been amalgamated and/or formatted correctly, or `BUILD.bazel` is out of date.' + (hasPatch ? '\n\n📎 A ready-to-apply patch is attached to the [failed workflow run](' + runUrl + ') as the `amalgamation-patch` artifact.' + ' Download it, then apply it locally from the repository root with:' + '\n\n```shell\ngit apply amalgamation.patch\n```\n\n' + 'This does not require installing astyle yourself.' : '') + (first ? '\n\n@' + author + ' Please read and follow the [Contribution Guidelines]' - + '(https://github.com/nlohmann/json/blob/develop/.github/CONTRIBUTING.md#files-to-change).' + + '(https://github.com/nlohmann/json/blob/develop/.github/CONTRIBUTING.md#amalgamate-the-source-code).' : '') }) diff --git a/BUILD.bazel b/BUILD.bazel index de0ff7145..ea8ffae21 100644 --- a/BUILD.bazel +++ b/BUILD.bazel @@ -30,8 +30,10 @@ cc_library( "include/nlohmann/detail/input/input_adapters.hpp", "include/nlohmann/detail/input/json_sax.hpp", "include/nlohmann/detail/input/lexer.hpp", + "include/nlohmann/detail/input/number_parse.hpp", "include/nlohmann/detail/input/parser.hpp", "include/nlohmann/detail/input/position_t.hpp", + "include/nlohmann/detail/input/string_scan.hpp", "include/nlohmann/detail/iterators/internal_iterator.hpp", "include/nlohmann/detail/iterators/iter_impl.hpp", "include/nlohmann/detail/iterators/iteration_proxy.hpp", @@ -49,6 +51,7 @@ cc_library( "include/nlohmann/detail/meta/detected.hpp", "include/nlohmann/detail/meta/identity_tag.hpp", "include/nlohmann/detail/meta/is_sax.hpp", + "include/nlohmann/detail/meta/logic.hpp", "include/nlohmann/detail/meta/std_fs.hpp", "include/nlohmann/detail/meta/type_traits.hpp", "include/nlohmann/detail/meta/void_t.hpp", diff --git a/FILES.md b/FILES.md index b68167336..263647146 100644 --- a/FILES.md +++ b/FILES.md @@ -250,12 +250,16 @@ Further documentation: ### `BUILD.bazel` -The file can be updated by calling +The build definition for [Bazel](https://bazel.build). The file is generated by +`cmake/scripts/gen_bazel_build_file.cmake`, which derives the header list from the files in `include`; change the +script rather than editing the file by hand. The file can be updated by calling ```shell make BUILD.bazel ``` +The "Check amalgamation" workflow fails if the file is out of date. + ### `meson.build` The build definition for the [Meson](https://mesonbuild.com) build system. diff --git a/Makefile b/Makefile index e1a1d2b75..871ea7995 100644 --- a/Makefile +++ b/Makefile @@ -1,4 +1,4 @@ -.PHONY: pretty clean ChangeLog.md release update_hedley update_hedley_undef +.PHONY: pretty clean ChangeLog.md release update_hedley update_hedley_undef BUILD.bazel ########################################################################## # configuration @@ -30,8 +30,9 @@ AMALGAMATED_FWD_FILE=single_include/nlohmann/json_fwd.hpp # main target all: @echo "amalgamate - amalgamate files single_include/nlohmann/json{,_fwd}.hpp from the include/nlohmann sources" + @echo "BUILD.bazel - regenerate the Bazel BUILD file from the include/nlohmann sources" @echo "ChangeLog.md - generate ChangeLog file" - @echo "check-amalgamation - check whether sources have been amalgamated" + @echo "check-amalgamation - check whether sources have been amalgamated and BUILD.bazel is up to date" @echo "clean - remove built files" @echo "doctest - compile example files and check their output" @echo "fuzz_testing - prepare fuzz testing of the JSON parser" @@ -172,8 +173,13 @@ check-amalgamation: @diff $(AMALGAMATED_FWD_FILE) $(AMALGAMATED_FWD_FILE)~ || (echo "===================================================================\n Amalgamation required! Please read the contribution guidelines\n in file .github/CONTRIBUTING.md.\n===================================================================" ; mv $(AMALGAMATED_FWD_FILE)~ $(AMALGAMATED_FWD_FILE) ; false) @mv $(AMALGAMATED_FILE)~ $(AMALGAMATED_FILE) @mv $(AMALGAMATED_FWD_FILE)~ $(AMALGAMATED_FWD_FILE) + @mv BUILD.bazel BUILD.bazel~ + @$(MAKE) BUILD.bazel + @diff BUILD.bazel BUILD.bazel~ || (echo "===================================================================\n BUILD.bazel is out of date! Please run 'make BUILD.bazel'.\n===================================================================" ; mv BUILD.bazel~ BUILD.bazel ; false) + @mv BUILD.bazel~ BUILD.bazel -BUILD.bazel: $(SRCS) +# generate the Bazel BUILD file; phony, because a removed header would not trigger a rebuild +BUILD.bazel: cmake -P cmake/scripts/gen_bazel_build_file.cmake ########################################################################## diff --git a/cmake/scripts/gen_bazel_build_file.cmake b/cmake/scripts/gen_bazel_build_file.cmake index e754d387d..3c7db9493 100644 --- a/cmake/scripts/gen_bazel_build_file.cmake +++ b/cmake/scripts/gen_bazel_build_file.cmake @@ -1,24 +1,58 @@ # generate Bazel BUILD file +# +# usage: cmake -P cmake/scripts/gen_bazel_build_file.cmake (or: make BUILD.bazel) +# +# The header list of the "json" target is derived from the files in include/. Everything else is fixed text below, +# so edit this script rather than BUILD.bazel. -set(PROJECT_ROOT "${CMAKE_CURRENT_LIST_DIR}/../..") +get_filename_component(PROJECT_ROOT "${CMAKE_CURRENT_LIST_DIR}/../.." ABSOLUTE) set(BUILD_FILE "${PROJECT_ROOT}/BUILD.bazel") -file(GLOB_RECURSE HEADERS LIST_DIRECTORIES false RELATIVE "${PROJECT_ROOT}" "include/*.hpp") +file(GLOB_RECURSE HEADERS LIST_DIRECTORIES false RELATIVE "${PROJECT_ROOT}" "${PROJECT_ROOT}/include/*.hpp") +list(SORT HEADERS) + +set(CONTENT [=[ +load("@rules_cc//cc:cc_library.bzl", "cc_library") +load("@rules_license//rules:license.bzl", "license") + +package( + default_applicable_licenses = [":license"], +) + +exports_files([ + "LICENSE.MIT", +]) + +license( + name = "license", + license_kinds = ["@rules_license//licenses/spdx:MIT"], + license_text = "LICENSE.MIT", +) -file(WRITE "${BUILD_FILE}" [=[ cc_library( name = "json", hdrs = [ ]=]) foreach(header ${HEADERS}) - file(APPEND "${BUILD_FILE}" " \"${header}\",\n") + string(APPEND CONTENT " \"${header}\",\n") endforeach() -file(APPEND "${BUILD_FILE}" [=[ +string(APPEND CONTENT [=[ ], includes = ["include"], visibility = ["//visibility:public"], alwayslink = True, ) + +cc_library( + name = "singleheader-json", + hdrs = [ + "single_include/nlohmann/json.hpp", + ], + includes = ["single_include"], + visibility = ["//visibility:public"], +) ]=]) + +file(WRITE "${BUILD_FILE}" "${CONTENT}") From 7c90ec2323fd85d8067961fb1ab808c6a0d1bac2 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Thu, 24 Sep 2026 17:12:02 +0200 Subject: [PATCH 32/64] Hash deeply nested values without recursing per nesting level (#5546) * Hash deeply nested values without recursing per nesting level std::hash hashed an array or object by hashing each element, which called detail::hash again once per nesting level. A value nested deeply enough - 50,000 levels of objects on an 8 MiB stack - exhausted the call stack and terminated the process. parse() accepts such values without complaint, since the parser is iterative, and a parsed value is hashed wherever it is used as a key in an unordered container. Bound the descent the same way dump() does: detail::hash takes the nesting level, and once hash_depth_limit() (128) levels have been entered, hash_iteratively() hashes what is left on an explicit stack. It combines the seeds in exactly the same order, so hash values are unchanged. A value nested less deeply than the bound is hashed by the same code as before, without allocating, and is as fast as before. Tests check that every depth up to twice the bound hashes exactly like the recursive definition of the hash, and that values nested 100,000 levels deep hash without crashing. Fixes #5545 for std::hash. Signed-off-by: Niels Lohmann * Declare hash_frame's constructor noexcept GCC's -Wnoexcept (an error in CI) flags the emplace_back() into the hash stack under C++26: the constructor cannot throw, since cbegin() is noexcept, but it did not say so. dump_frame's constructor is noexcept for the same reason. Signed-off-by: Niels Lohmann * Share one recursion depth limit, and copy the hash frame out of the stack dump() and hash() each defined their own limit on how many nesting levels they recurse into, and the operations still to come would have added more, free to diverge over time. They now all use detail::recursion_depth_limit(), in a header of its own; serializer::dump_depth_limit() and hash_depth_limit() are gone. hash_iteratively() now copies the frame it works on out of the stack and changes the frame only through stack.back(), so nothing can refer into the stack after entering an element has grown it. Signed-off-by: Niels Lohmann * Parenthesize multiplications in the hash test for clang-tidy Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- BUILD.bazel | 1 + include/nlohmann/detail/hash.hpp | 102 ++++++++++- include/nlohmann/detail/output/serializer.hpp | 16 +- .../nlohmann/detail/recursion_depth_limit.hpp | 35 ++++ include/nlohmann/json.hpp | 1 + single_include/nlohmann/json.hpp | 158 ++++++++++++++++-- tests/src/unit-hash.cpp | 113 +++++++++++++ 7 files changed, 398 insertions(+), 28 deletions(-) create mode 100644 include/nlohmann/detail/recursion_depth_limit.hpp diff --git a/BUILD.bazel b/BUILD.bazel index ea8ffae21..b13e62c22 100644 --- a/BUILD.bazel +++ b/BUILD.bazel @@ -58,6 +58,7 @@ cc_library( "include/nlohmann/detail/output/binary_writer.hpp", "include/nlohmann/detail/output/output_adapters.hpp", "include/nlohmann/detail/output/serializer.hpp", + "include/nlohmann/detail/recursion_depth_limit.hpp", "include/nlohmann/detail/string_concat.hpp", "include/nlohmann/detail/string_escape.hpp", "include/nlohmann/detail/string_utils.hpp", diff --git a/include/nlohmann/detail/hash.hpp b/include/nlohmann/detail/hash.hpp index be8063f89..20a971886 100644 --- a/include/nlohmann/detail/hash.hpp +++ b/include/nlohmann/detail/hash.hpp @@ -11,8 +11,10 @@ #include // uint8_t #include // size_t #include // hash +#include // vector #include +#include #include NLOHMANN_JSON_NAMESPACE_BEGIN @@ -26,6 +28,9 @@ inline std::size_t combine(std::size_t seed, std::size_t h) noexcept return seed; } +template +std::size_t hash_iteratively(const BasicJsonType& j); + /*! @brief hash a JSON value @@ -33,12 +38,21 @@ The hash function tries to rely on std::hash where possible. Furthermore, the type of the JSON value is taken into account to have different hash values for null, 0, 0U, and false, etc. +Hashing an array or an object hashes its elements, which used to call this +function again once per nesting level, so a value nested deeply enough +exhausted the call stack and terminated the process. The descent is bounded +here: once @ref recursion_depth_limit levels have been entered, @ref +hash_iteratively hashes what is left without the call stack. A value nested +less deeply than that - all but a vanishing minority - is hashed exactly as +before, without allocating. + @tparam BasicJsonType basic_json specialization @param j JSON value to hash +@param depth nesting level of @a j, counted from the value passed by the caller @return hash value of j */ template -std::size_t hash(const BasicJsonType& j) +std::size_t hash(const BasicJsonType& j, const std::size_t depth = 0) { using string_t = typename BasicJsonType::string_t; using number_integer_t = typename BasicJsonType::number_integer_t; @@ -56,22 +70,32 @@ std::size_t hash(const BasicJsonType& j) case BasicJsonType::value_t::object: { + if (JSON_HEDLEY_UNLIKELY(depth >= recursion_depth_limit())) + { + return hash_iteratively(j); + } + auto seed = combine(type, j.size()); for (const auto& element : j.items()) { const auto h = std::hash {}(element.key()); seed = combine(seed, h); - seed = combine(seed, hash(element.value())); + seed = combine(seed, hash(element.value(), depth + 1)); } return seed; } case BasicJsonType::value_t::array: { + if (JSON_HEDLEY_UNLIKELY(depth >= recursion_depth_limit())) + { + return hash_iteratively(j); + } + auto seed = combine(type, j.size()); for (const auto& element : j) { - seed = combine(seed, hash(element)); + seed = combine(seed, hash(element, depth + 1)); } return seed; } @@ -127,5 +151,77 @@ std::size_t hash(const BasicJsonType& j) } } +/// an array or object whose elements @ref hash_iteratively is hashing +template +struct hash_frame +{ + hash_frame(const BasicJsonType* value_, std::size_t seed_) noexcept + : value(value_), position(value_->cbegin()), seed(seed_) + {} + + const BasicJsonType* value; + typename BasicJsonType::const_iterator position; + std::size_t seed; +}; + +/*! +@brief hash the array or object @a j without the call stack + +Computes the same value as @ref hash, keeping the arrays and objects it has +entered on an explicit stack instead of descending into them. Only reached for +values nested deeper than @ref recursion_depth_limit. + +@tparam BasicJsonType basic_json specialization +@param j array or object to hash +@return hash value of j +*/ +template +std::size_t hash_iteratively(const BasicJsonType& j) +{ + using string_t = typename BasicJsonType::string_t; + + std::vector> stack; + stack.emplace_back(&j, combine(static_cast(j.type()), j.size())); + + while (true) + { + // a copy, as entering an element below can reallocate the stack; the + // frame itself is only changed through stack.back() + const hash_frame frame = stack.back(); + + if (frame.position == frame.value->cend()) + { + // all elements are hashed: fold this value's hash into its parent's + // seed, exactly where the recursive version returns it + const std::size_t h = frame.seed; + stack.pop_back(); + if (stack.empty()) + { + return h; + } + stack.back().seed = combine(stack.back().seed, h); + continue; + } + + if (frame.value->is_object()) + { + stack.back().seed = combine(stack.back().seed, std::hash {}(frame.position.key())); + } + + // advance before entering the element, which pushes onto the stack + const BasicJsonType& element = *frame.position; + ++stack.back().position; + + if (element.is_structured()) + { + stack.emplace_back(&element, combine(static_cast(element.type()), element.size())); + } + else + { + stack.back().seed = combine(stack.back().seed, hash(element)); + } + } +} + } // namespace detail NLOHMANN_JSON_NAMESPACE_END diff --git a/include/nlohmann/detail/output/serializer.hpp b/include/nlohmann/detail/output/serializer.hpp index f9e7f7840..7c38276ce 100644 --- a/include/nlohmann/detail/output/serializer.hpp +++ b/include/nlohmann/detail/output/serializer.hpp @@ -30,6 +30,7 @@ #include #include #include +#include #include #include @@ -133,7 +134,7 @@ class serializer Serializing a container descends into its elements, so a value nested deeply enough used to exhaust the call stack and terminate the process with no - exception to catch. The descent is bounded here: once @ref dump_depth_limit + exception to catch. The descent is bounded here: once @ref recursion_depth_limit levels have been entered, @ref dump_iteratively writes out what is left without the call stack. A value nested less deeply than that - all but a vanishing minority - is written by exactly the code that always wrote it. @@ -148,7 +149,7 @@ class serializer { case value_t::object: { - if (JSON_HEDLEY_UNLIKELY(depth >= dump_depth_limit())) + if (JSON_HEDLEY_UNLIKELY(depth >= recursion_depth_limit())) { dump_iteratively(val, current_indent); return; @@ -223,7 +224,7 @@ class serializer case value_t::array: { - if (JSON_HEDLEY_UNLIKELY(depth >= dump_depth_limit())) + if (JSON_HEDLEY_UNLIKELY(depth >= recursion_depth_limit())) { dump_iteratively(val, current_indent); return; @@ -408,19 +409,12 @@ class serializer } private: - /// the number of levels @ref dump_internal descends into before it hands - /// over to @ref dump_iteratively - static constexpr std::size_t dump_depth_limit() - { - return 128; - } - /*! @brief write out @a val and everything below it without the call stack Emits the same bytes as @ref dump_internal, keeping the containers it has entered on an explicit stack instead of descending into them. Only reached - for values nested deeper than @ref dump_depth_limit, which is why it is not + for values nested deeper than @ref recursion_depth_limit, which is why it is not written for speed: walking every value this way measured up to 20% slower on object-heavy documents than letting the compiler drive the descent. */ diff --git a/include/nlohmann/detail/recursion_depth_limit.hpp b/include/nlohmann/detail/recursion_depth_limit.hpp new file mode 100644 index 000000000..fe3bd8026 --- /dev/null +++ b/include/nlohmann/detail/recursion_depth_limit.hpp @@ -0,0 +1,35 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +#include // size_t + +#include + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ + +/*! +@brief the number of nesting levels an operation recurses into + +Operations that walk a value (serializing, hashing, merging, ...) recurse once +per nesting level, which is fastest, but a value nested deeply enough would +exhaust the call stack. So they recurse only this many levels deep and finish +whatever lies below with an explicit stack. All of them share this limit. + +@sa https://github.com/nlohmann/json/issues/5387 +*/ +constexpr std::size_t recursion_depth_limit() noexcept +{ + return 128; +} + +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END diff --git a/include/nlohmann/json.hpp b/include/nlohmann/json.hpp index 9c1d821e1..b64feb4a9 100644 --- a/include/nlohmann/json.hpp +++ b/include/nlohmann/json.hpp @@ -68,6 +68,7 @@ #include #include #include +#include #include #include #include diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index c11fa27e5..335a29289 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -7028,9 +7028,48 @@ NLOHMANN_JSON_NAMESPACE_END #include // uint8_t #include // size_t #include // hash +#include // vector // #include +// #include +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + + + +#include // size_t + +// #include + + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ + +/*! +@brief the number of nesting levels an operation recurses into + +Operations that walk a value (serializing, hashing, merging, ...) recurse once +per nesting level, which is fastest, but a value nested deeply enough would +exhaust the call stack. So they recurse only this many levels deep and finish +whatever lies below with an explicit stack. All of them share this limit. + +@sa https://github.com/nlohmann/json/issues/5387 +*/ +constexpr std::size_t recursion_depth_limit() noexcept +{ + return 128; +} + +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END + // #include @@ -7045,6 +7084,9 @@ inline std::size_t combine(std::size_t seed, std::size_t h) noexcept return seed; } +template +std::size_t hash_iteratively(const BasicJsonType& j); + /*! @brief hash a JSON value @@ -7052,12 +7094,21 @@ The hash function tries to rely on std::hash where possible. Furthermore, the type of the JSON value is taken into account to have different hash values for null, 0, 0U, and false, etc. +Hashing an array or an object hashes its elements, which used to call this +function again once per nesting level, so a value nested deeply enough +exhausted the call stack and terminated the process. The descent is bounded +here: once @ref recursion_depth_limit levels have been entered, @ref +hash_iteratively hashes what is left without the call stack. A value nested +less deeply than that - all but a vanishing minority - is hashed exactly as +before, without allocating. + @tparam BasicJsonType basic_json specialization @param j JSON value to hash +@param depth nesting level of @a j, counted from the value passed by the caller @return hash value of j */ template -std::size_t hash(const BasicJsonType& j) +std::size_t hash(const BasicJsonType& j, const std::size_t depth = 0) { using string_t = typename BasicJsonType::string_t; using number_integer_t = typename BasicJsonType::number_integer_t; @@ -7075,22 +7126,32 @@ std::size_t hash(const BasicJsonType& j) case BasicJsonType::value_t::object: { + if (JSON_HEDLEY_UNLIKELY(depth >= recursion_depth_limit())) + { + return hash_iteratively(j); + } + auto seed = combine(type, j.size()); for (const auto& element : j.items()) { const auto h = std::hash {}(element.key()); seed = combine(seed, h); - seed = combine(seed, hash(element.value())); + seed = combine(seed, hash(element.value(), depth + 1)); } return seed; } case BasicJsonType::value_t::array: { + if (JSON_HEDLEY_UNLIKELY(depth >= recursion_depth_limit())) + { + return hash_iteratively(j); + } + auto seed = combine(type, j.size()); for (const auto& element : j) { - seed = combine(seed, hash(element)); + seed = combine(seed, hash(element, depth + 1)); } return seed; } @@ -7146,6 +7207,78 @@ std::size_t hash(const BasicJsonType& j) } } +/// an array or object whose elements @ref hash_iteratively is hashing +template +struct hash_frame +{ + hash_frame(const BasicJsonType* value_, std::size_t seed_) noexcept + : value(value_), position(value_->cbegin()), seed(seed_) + {} + + const BasicJsonType* value; + typename BasicJsonType::const_iterator position; + std::size_t seed; +}; + +/*! +@brief hash the array or object @a j without the call stack + +Computes the same value as @ref hash, keeping the arrays and objects it has +entered on an explicit stack instead of descending into them. Only reached for +values nested deeper than @ref recursion_depth_limit. + +@tparam BasicJsonType basic_json specialization +@param j array or object to hash +@return hash value of j +*/ +template +std::size_t hash_iteratively(const BasicJsonType& j) +{ + using string_t = typename BasicJsonType::string_t; + + std::vector> stack; + stack.emplace_back(&j, combine(static_cast(j.type()), j.size())); + + while (true) + { + // a copy, as entering an element below can reallocate the stack; the + // frame itself is only changed through stack.back() + const hash_frame frame = stack.back(); + + if (frame.position == frame.value->cend()) + { + // all elements are hashed: fold this value's hash into its parent's + // seed, exactly where the recursive version returns it + const std::size_t h = frame.seed; + stack.pop_back(); + if (stack.empty()) + { + return h; + } + stack.back().seed = combine(stack.back().seed, h); + continue; + } + + if (frame.value->is_object()) + { + stack.back().seed = combine(stack.back().seed, std::hash {}(frame.position.key())); + } + + // advance before entering the element, which pushes onto the stack + const BasicJsonType& element = *frame.position; + ++stack.back().position; + + if (element.is_structured()) + { + stack.emplace_back(&element, combine(static_cast(element.type()), element.size())); + } + else + { + stack.back().seed = combine(stack.back().seed, hash(element)); + } + } +} + } // namespace detail NLOHMANN_JSON_NAMESPACE_END @@ -22295,6 +22428,8 @@ NLOHMANN_JSON_NAMESPACE_END // #include +// #include + // #include // #include @@ -22400,7 +22535,7 @@ class serializer Serializing a container descends into its elements, so a value nested deeply enough used to exhaust the call stack and terminate the process with no - exception to catch. The descent is bounded here: once @ref dump_depth_limit + exception to catch. The descent is bounded here: once @ref recursion_depth_limit levels have been entered, @ref dump_iteratively writes out what is left without the call stack. A value nested less deeply than that - all but a vanishing minority - is written by exactly the code that always wrote it. @@ -22415,7 +22550,7 @@ class serializer { case value_t::object: { - if (JSON_HEDLEY_UNLIKELY(depth >= dump_depth_limit())) + if (JSON_HEDLEY_UNLIKELY(depth >= recursion_depth_limit())) { dump_iteratively(val, current_indent); return; @@ -22490,7 +22625,7 @@ class serializer case value_t::array: { - if (JSON_HEDLEY_UNLIKELY(depth >= dump_depth_limit())) + if (JSON_HEDLEY_UNLIKELY(depth >= recursion_depth_limit())) { dump_iteratively(val, current_indent); return; @@ -22675,19 +22810,12 @@ class serializer } private: - /// the number of levels @ref dump_internal descends into before it hands - /// over to @ref dump_iteratively - static constexpr std::size_t dump_depth_limit() - { - return 128; - } - /*! @brief write out @a val and everything below it without the call stack Emits the same bytes as @ref dump_internal, keeping the containers it has entered on an explicit stack instead of descending into them. Only reached - for values nested deeper than @ref dump_depth_limit, which is why it is not + for values nested deeper than @ref recursion_depth_limit, which is why it is not written for speed: walking every value this way measured up to 20% slower on object-heavy documents than letting the compiler drive the descent. */ @@ -24000,6 +24128,8 @@ class serializer } // namespace detail NLOHMANN_JSON_NAMESPACE_END +// #include + // #include // #include diff --git a/tests/src/unit-hash.cpp b/tests/src/unit-hash.cpp index c161efa6e..eb843c291 100644 --- a/tests/src/unit-hash.cpp +++ b/tests/src/unit-hash.cpp @@ -13,6 +13,78 @@ using json = nlohmann::json; using ordered_json = nlohmann::ordered_json; #include +#include + +namespace +{ +// how detail::hash defines the hash of an array or object: the seeds of the +// elements, combined in order. Recursive, so only usable on values nested a +// few hundred levels deep - which is exactly what is needed to check that the +// iterative path taken below detail::recursion_depth_limit() computes the same. +template +std::size_t reference_hash(const BasicJsonType& j) +{ + using nlohmann::detail::combine; + using string_t = typename BasicJsonType::string_t; + + if (!j.is_structured()) + { + return std::hash {}(j); + } + + auto seed = combine(static_cast(j.type()), j.size()); + for (const auto& element : j.items()) + { + if (j.is_object()) + { + seed = combine(seed, std::hash {}(element.key())); + } + seed = combine(seed, reference_hash(element.value())); + } + return seed; +} + +// a value nested `depth` levels deep, with siblings on every level +template +BasicJsonType nested(const std::size_t depth, const bool objects) +{ + BasicJsonType value = "leaf"; + for (std::size_t i = 0; i < depth; ++i) + { + if (objects) + { + value = BasicJsonType{{"before", i}, {"nested", std::move(value)}, {"after", {i, "x"}}}; + } + else + { + value = BasicJsonType::array({i, std::move(value), BasicJsonType::object({{"k", i}})}); + } + } + return value; +} + +std::string nested_text(const std::size_t depth, const bool objects) +{ + std::string text; + if (objects) + { + text.reserve((6 * depth) + 1); + for (std::size_t i = 0; i < depth; ++i) + { + text += "{\"a\":"; + } + text += "1"; + text.append(depth, '}'); + } + else + { + text.assign(depth, '['); + text += "1"; + text.append(depth, ']'); + } + return text; +} +} // namespace TEST_CASE("hash") { @@ -111,3 +183,44 @@ TEST_CASE("hash") CHECK(hashes.size() == 21); } + +TEST_CASE("hash of deeply nested values") +{ + SECTION("hashing past the descent bound computes the same values") + { + // every depth on either side of where the iterative path takes over + for (std::size_t depth = 0; depth <= (2 * nlohmann::detail::recursion_depth_limit()) + 10; ++depth) + { + CAPTURE(depth); + const auto arrays = nested(depth, false); + const auto objects = nested(depth, true); + const auto ordered = nested(depth, true); + CHECK(std::hash {}(arrays) == reference_hash(arrays)); + CHECK(std::hash {}(objects) == reference_hash(objects)); + CHECK(std::hash {}(ordered) == reference_hash(ordered)); + } + } + + SECTION("values nested too deeply for the call stack (#5545)") + { + // recursing once per level used to exhaust the call stack here; the + // values are only parsed and hashed, never copied or compared, since + // those recurse as well + const std::size_t depth = 100000; + for (const bool objects : + { + false, true + }) + { + CAPTURE(objects); + const auto text = nested_text(depth, objects); + const auto a = json::parse(text); + const auto b = json::parse(text); + CHECK(std::hash {}(a) == std::hash {}(b)); + + const auto c = ordered_json::parse(text); + const auto d = ordered_json::parse(text); + CHECK(std::hash {}(c) == std::hash {}(d)); + } + } +} From 4daca40d7b03d05d1af730003f826c1ff91a90e8 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Thu, 24 Sep 2026 17:12:03 +0200 Subject: [PATCH 33/64] Merge deeply nested objects without recursing per nesting level (#5547) * Merge deeply nested objects without recursing per nesting level merge_patch() and update(j, true) merged a nested object by calling themselves on it, once per nesting level. A value nested deeply enough - 50,000 levels of objects on an 8 MiB stack - exhausted the call stack and terminated the process, although parse() accepts such values without complaint. Bound the descent the same way dump() does. The recursion now carries the nesting level, and once merge_depth_limit() (128) levels have been entered, update_members_iteratively() and merge_patch_iteratively() finish the merge on an explicit stack. They still merge a nested object completely before the next member, and in the same order, so the results, including the parents JSON_DIAGNOSTICS reports paths from, are unchanged. Values nested less deeply than the bound run the same code as before, so the common case does not pay for the stack: merging only on it cost 10-14% in a first version. The public signatures are unchanged. The recursive worker behind merge_patch() has its own name rather than being a private overload, so that &basic_json::merge_patch stays unambiguous. Tests check every depth up to 300 against recursive reference implementations of both operations, check the diagnostic paths past the bound, and merge objects nested 100,000 levels deep. Fixes #5545 for update(j, true), and #5393 for merge_patch(). Signed-off-by: Niels Lohmann * Use the shared recursion limit in update() and merge_patch() merge_depth_limit() is gone in favor of detail::recursion_depth_limit(). The two identical function-local frame structs become one member struct, merge_frame, with a constructor, so both loops emplace_back() their frames. merge_patch_iteratively() copies the frame it works on out of the stack and changes it only through stack.back(). Signed-off-by: Niels Lohmann * Build the update()/merge_patch() diagnostics test values instead of parsing them Parsed values carry byte positions under JSON_DIAGNOSTIC_POSITIONS, which the expected messages do not include. Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- include/nlohmann/json.hpp | 178 ++++++++++++++++++++++++++++++- single_include/nlohmann/json.hpp | 178 ++++++++++++++++++++++++++++++- tests/src/unit-diagnostics.cpp | 49 +++++++++ tests/src/unit-merge_patch.cpp | 103 ++++++++++++++++++ tests/src/unit-modifiers.cpp | 88 +++++++++++++++ 5 files changed, 590 insertions(+), 6 deletions(-) diff --git a/include/nlohmann/json.hpp b/include/nlohmann/json.hpp index b64feb4a9..e40a2aa27 100644 --- a/include/nlohmann/json.hpp +++ b/include/nlohmann/json.hpp @@ -3913,17 +3913,54 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_THROW(type_error::create(312, detail::concat("cannot use update() with ", first.m_object->type_name()), first.m_object)); } + update_members(first, last, merge_objects, 0); + } + + private: + /// @brief an object @ref update_members_iteratively or @ref + /// merge_patch_iteratively is merging into, and the members still to merge + struct merge_frame + { + merge_frame(basic_json* target_, const_iterator position_, const_iterator last_) noexcept + : target(target_), position(std::move(position_)), last(std::move(last_)) + {} + + basic_json* target; + const_iterator position; + const_iterator last; + }; + + /*! + @brief the members loop of @ref update, for this object and range + + Merging a nested object calls this function again, once per nesting + level, so a value nested deeply enough used to exhaust the call stack and + terminate the process. The descent is bounded here: once @ref + detail::recursion_depth_limit levels have been entered, @ref + update_members_iteratively merges what is left without the call stack. + + @param[in] depth nesting level of this object, counted from the object + @ref update was called on + */ + void update_members(const const_iterator& first, const const_iterator& last, const bool merge_objects, const std::size_t depth) + { + if (JSON_HEDLEY_UNLIKELY(depth >= detail::recursion_depth_limit())) + { + update_members_iteratively(first, last); + return; + } + for (auto it = first; it != last; ++it) { if (merge_objects && it.value().is_object()) { - auto it2 = m_data.m_value.object->find(it.key()); + const auto it2 = m_data.m_value.object->find(it.key()); // Only recurse when the existing value is itself an object. // Otherwise overwrite, matching the documented "all other values // are overwritten as usual" behavior (see #5402). if (it2 != m_data.m_value.object->end() && it2->second.is_object()) { - it2->second.update(it.value(), true); + it2->second.update_members(it.value().cbegin(), it.value().cend(), true, depth + 1); #if JSON_DIAGNOSTICS it2->second.set_parents(); #endif @@ -3937,6 +3974,64 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } } + /*! + @brief merge @a first to @a last into this object without the call stack + + Does the same as @ref update_members with `merge_objects` set, keeping the + objects whose merge was interrupted by a nested one on an explicit stack + instead of descending into them. A nested object is still merged + completely before the next member, in the same order as the recursive + version. Only reached for values nested deeper than @ref + detail::recursion_depth_limit. + */ + void update_members_iteratively(const_iterator first, const_iterator last) + { + std::vector stack; + + basic_json* target = this; + while (true) + { + if (first == last) + { + if (stack.empty()) + { + break; + } + + // a nested object is merged: continue with its parent +#if JSON_DIAGNOSTICS + target->set_parents(); +#endif + target = stack.back().target; + first = stack.back().position; + last = stack.back().last; + stack.pop_back(); + continue; + } + + if (first.value().is_object()) + { + const auto it2 = target->m_data.m_value.object->find(first.key()); + if (it2 != target->m_data.m_value.object->end() && it2->second.is_object()) + { + const basic_json& source = first.value(); + ++first; + stack.emplace_back(target, first, last); + target = &it2->second; + first = source.cbegin(); + last = source.cend(); + continue; + } + } + target->m_data.m_value.object->operator[](first.key()) = first.value(); +#if JSON_DIAGNOSTICS + target->m_data.m_value.object->operator[](first.key()).m_parent = target; +#endif + ++first; + } + } + + public: /// @brief exchanges the values /// @sa https://json.nlohmann.me/api/basic_json/swap/ void swap(reference other) noexcept ( @@ -5831,9 +5926,30 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @brief applies a JSON Merge Patch /// @sa https://json.nlohmann.me/api/basic_json/merge_patch/ void merge_patch(const basic_json& apply_patch) + { + apply_merge_patch(apply_patch, 0); + } + + private: + /*! + @brief @ref merge_patch, for a patch at nesting level @a depth + + Applying a nested object calls this function again, once per nesting + level, so a patch nested deeply enough used to exhaust the call stack and + terminate the process. The descent is bounded here: once @ref + detail::recursion_depth_limit levels have been entered, @ref + merge_patch_iteratively applies what is left without the call stack. + */ + void apply_merge_patch(const basic_json& apply_patch, const std::size_t depth) { if (apply_patch.is_object()) { + if (JSON_HEDLEY_UNLIKELY(depth >= detail::recursion_depth_limit())) + { + merge_patch_iteratively(apply_patch); + return; + } + if (!is_object()) { *this = object(); @@ -5846,7 +5962,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } else { - operator[](it.key()).merge_patch(it.value()); + operator[](it.key()).apply_merge_patch(it.value(), depth + 1); } } } @@ -5856,6 +5972,62 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } } + /*! + @brief apply @a apply_patch to this value without the call stack + + Does the same as @ref merge_patch, keeping the objects being patched on an + explicit stack instead of descending into them. A nested object is still + patched completely before the next member, in the same order as the + recursive version. Only reached for patches nested deeper than @ref + detail::recursion_depth_limit. + */ + void merge_patch_iteratively(const basic_json& apply_patch) + { + std::vector stack; + + // patch `target` with `patch`, or start patching it member by member + const auto apply = [&stack](basic_json & target, const basic_json & patch) + { + if (patch.is_object()) + { + if (!target.is_object()) + { + target = basic_json::object(); + } + stack.emplace_back(&target, patch.cbegin(), patch.cend()); + } + else + { + target = patch; + } + }; + + apply(*this, apply_patch); + while (!stack.empty()) + { + // a copy, as applying a member below can reallocate the stack; + // the frame itself is only changed through stack.back() + const merge_frame frame = stack.back(); + if (frame.position == frame.last) + { + stack.pop_back(); + continue; + } + + const const_iterator member = frame.position; + ++stack.back().position; + if (member.value().is_null()) + { + frame.target->erase(member.key()); + } + else + { + apply(frame.target->operator[](member.key()), member.value()); + } + } + } + + public: /// @} }; diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 335a29289..3c5821d92 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -28371,17 +28371,54 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_THROW(type_error::create(312, detail::concat("cannot use update() with ", first.m_object->type_name()), first.m_object)); } + update_members(first, last, merge_objects, 0); + } + + private: + /// @brief an object @ref update_members_iteratively or @ref + /// merge_patch_iteratively is merging into, and the members still to merge + struct merge_frame + { + merge_frame(basic_json* target_, const_iterator position_, const_iterator last_) noexcept + : target(target_), position(std::move(position_)), last(std::move(last_)) + {} + + basic_json* target; + const_iterator position; + const_iterator last; + }; + + /*! + @brief the members loop of @ref update, for this object and range + + Merging a nested object calls this function again, once per nesting + level, so a value nested deeply enough used to exhaust the call stack and + terminate the process. The descent is bounded here: once @ref + detail::recursion_depth_limit levels have been entered, @ref + update_members_iteratively merges what is left without the call stack. + + @param[in] depth nesting level of this object, counted from the object + @ref update was called on + */ + void update_members(const const_iterator& first, const const_iterator& last, const bool merge_objects, const std::size_t depth) + { + if (JSON_HEDLEY_UNLIKELY(depth >= detail::recursion_depth_limit())) + { + update_members_iteratively(first, last); + return; + } + for (auto it = first; it != last; ++it) { if (merge_objects && it.value().is_object()) { - auto it2 = m_data.m_value.object->find(it.key()); + const auto it2 = m_data.m_value.object->find(it.key()); // Only recurse when the existing value is itself an object. // Otherwise overwrite, matching the documented "all other values // are overwritten as usual" behavior (see #5402). if (it2 != m_data.m_value.object->end() && it2->second.is_object()) { - it2->second.update(it.value(), true); + it2->second.update_members(it.value().cbegin(), it.value().cend(), true, depth + 1); #if JSON_DIAGNOSTICS it2->second.set_parents(); #endif @@ -28395,6 +28432,64 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } } + /*! + @brief merge @a first to @a last into this object without the call stack + + Does the same as @ref update_members with `merge_objects` set, keeping the + objects whose merge was interrupted by a nested one on an explicit stack + instead of descending into them. A nested object is still merged + completely before the next member, in the same order as the recursive + version. Only reached for values nested deeper than @ref + detail::recursion_depth_limit. + */ + void update_members_iteratively(const_iterator first, const_iterator last) + { + std::vector stack; + + basic_json* target = this; + while (true) + { + if (first == last) + { + if (stack.empty()) + { + break; + } + + // a nested object is merged: continue with its parent +#if JSON_DIAGNOSTICS + target->set_parents(); +#endif + target = stack.back().target; + first = stack.back().position; + last = stack.back().last; + stack.pop_back(); + continue; + } + + if (first.value().is_object()) + { + const auto it2 = target->m_data.m_value.object->find(first.key()); + if (it2 != target->m_data.m_value.object->end() && it2->second.is_object()) + { + const basic_json& source = first.value(); + ++first; + stack.emplace_back(target, first, last); + target = &it2->second; + first = source.cbegin(); + last = source.cend(); + continue; + } + } + target->m_data.m_value.object->operator[](first.key()) = first.value(); +#if JSON_DIAGNOSTICS + target->m_data.m_value.object->operator[](first.key()).m_parent = target; +#endif + ++first; + } + } + + public: /// @brief exchanges the values /// @sa https://json.nlohmann.me/api/basic_json/swap/ void swap(reference other) noexcept ( @@ -30289,9 +30384,30 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec /// @brief applies a JSON Merge Patch /// @sa https://json.nlohmann.me/api/basic_json/merge_patch/ void merge_patch(const basic_json& apply_patch) + { + apply_merge_patch(apply_patch, 0); + } + + private: + /*! + @brief @ref merge_patch, for a patch at nesting level @a depth + + Applying a nested object calls this function again, once per nesting + level, so a patch nested deeply enough used to exhaust the call stack and + terminate the process. The descent is bounded here: once @ref + detail::recursion_depth_limit levels have been entered, @ref + merge_patch_iteratively applies what is left without the call stack. + */ + void apply_merge_patch(const basic_json& apply_patch, const std::size_t depth) { if (apply_patch.is_object()) { + if (JSON_HEDLEY_UNLIKELY(depth >= detail::recursion_depth_limit())) + { + merge_patch_iteratively(apply_patch); + return; + } + if (!is_object()) { *this = object(); @@ -30304,7 +30420,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } else { - operator[](it.key()).merge_patch(it.value()); + operator[](it.key()).apply_merge_patch(it.value(), depth + 1); } } } @@ -30314,6 +30430,62 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } } + /*! + @brief apply @a apply_patch to this value without the call stack + + Does the same as @ref merge_patch, keeping the objects being patched on an + explicit stack instead of descending into them. A nested object is still + patched completely before the next member, in the same order as the + recursive version. Only reached for patches nested deeper than @ref + detail::recursion_depth_limit. + */ + void merge_patch_iteratively(const basic_json& apply_patch) + { + std::vector stack; + + // patch `target` with `patch`, or start patching it member by member + const auto apply = [&stack](basic_json & target, const basic_json & patch) + { + if (patch.is_object()) + { + if (!target.is_object()) + { + target = basic_json::object(); + } + stack.emplace_back(&target, patch.cbegin(), patch.cend()); + } + else + { + target = patch; + } + }; + + apply(*this, apply_patch); + while (!stack.empty()) + { + // a copy, as applying a member below can reallocate the stack; + // the frame itself is only changed through stack.back() + const merge_frame frame = stack.back(); + if (frame.position == frame.last) + { + stack.pop_back(); + continue; + } + + const const_iterator member = frame.position; + ++stack.back().position; + if (member.value().is_null()) + { + frame.target->erase(member.key()); + } + else + { + apply(frame.target->operator[](member.key()), member.value()); + } + } + } + + public: /// @} }; diff --git a/tests/src/unit-diagnostics.cpp b/tests/src/unit-diagnostics.cpp index 389a7a3d7..cdb6185b3 100644 --- a/tests/src/unit-diagnostics.cpp +++ b/tests/src/unit-diagnostics.cpp @@ -363,3 +363,52 @@ TEST_CASE("Regression tests for extended diagnostics") } } +TEST_CASE("Better diagnostics past the descent bound of update() and merge_patch()") +{ + // Both merge objects nested more than detail::recursion_depth_limit() + // (128) levels deep without recursing; the values they add or replace + // there must still know their parents. + // The values are built rather than parsed, so that the expected messages + // carry no byte positions under JSON_DIAGNOSTIC_POSITIONS. + const std::size_t depth = 200; + json target = {{"x", 1}}; + json patch = {{"y", 2}}; + std::string path; + for (std::size_t i = 0; i < depth; ++i) + { + target = json{{"a", std::move(target)}}; + patch = json{{"a", std::move(patch)}}; + path += "/a"; + } + const std::string expected_x = "[json.exception.type_error.304] (" + path + "/x) cannot use at() with number"; + const std::string expected_y = "[json.exception.type_error.304] (" + path + "/y) cannot use at() with number"; + + SECTION("update()") + { + json j = target; + j.update(patch, true); + + // walk down through const references, which leave m_parent alone + const json* p = &j; + for (std::size_t i = 0; i < depth; ++i) + { + p = &p->at("a"); + } + CHECK_THROWS_WITH_AS(p->at("x").at(0), expected_x.c_str(), json::type_error); + CHECK_THROWS_WITH_AS(p->at("y").at(0), expected_y.c_str(), json::type_error); + } + + SECTION("merge_patch()") + { + json j = target; + j.merge_patch(patch); + + const json* p = &j; + for (std::size_t i = 0; i < depth; ++i) + { + p = &p->at("a"); + } + CHECK_THROWS_WITH_AS(p->at("x").at(0), expected_x.c_str(), json::type_error); + CHECK_THROWS_WITH_AS(p->at("y").at(0), expected_y.c_str(), json::type_error); + } +} diff --git a/tests/src/unit-merge_patch.cpp b/tests/src/unit-merge_patch.cpp index f02a1e991..8ac281ef5 100644 --- a/tests/src/unit-merge_patch.cpp +++ b/tests/src/unit-merge_patch.cpp @@ -14,6 +14,60 @@ using nlohmann::json; using namespace nlohmann::literals; // NOLINT(google-build-using-namespace) #endif +#include + +namespace +{ +// RFC 7396's MergePatch, written recursively as in the RFC; only usable on +// values nested a few hundred levels deep +void reference_merge_patch(json& target, const json& patch) +{ + if (!patch.is_object()) + { + target = patch; + return; + } + if (!target.is_object()) + { + target = json::object(); + } + for (auto it = patch.begin(); it != patch.end(); ++it) + { + if (it.value().is_null()) + { + target.erase(it.key()); + } + else + { + reference_merge_patch(target[it.key()], it.value()); + } + } +} + +// objects nested `depth` levels deep under the key "a", with members that +// differ by `variant` on the way down +std::string nested_objects(const std::size_t depth, const int variant) +{ + std::string text; + for (std::size_t i = 0; i < depth; ++i) + { + text += "{"; + if ((i + static_cast(variant)) % 3 == 0) + { + text += "\"s" + std::to_string(variant) + "\":" + std::to_string(i) + ","; + } + if (variant == 2 && i % 5 == 0) + { + text += "\"s0\":null,"; + } + text += "\"a\":"; + } + text += variant == 1 ? "{\"x\":1,\"y\":null}" : "{\"y\":2}"; + text.append(depth, '}'); + return text; +} +} // namespace + TEST_CASE("JSON Merge Patch") { SECTION("examples from RFC 7396") @@ -242,3 +296,52 @@ TEST_CASE("JSON Merge Patch") } } } + +TEST_CASE("JSON Merge Patch on deeply nested values") +{ + SECTION("patching past the descent bound gives the same result") + { + // every depth on either side of where the iterative version takes + // over (detail::recursion_depth_limit(), 128) + for (std::size_t depth = 0; depth <= 300; ++depth) + { + CAPTURE(depth); + for (int variant = 0; variant < 3; ++variant) + { + CAPTURE(variant); + const json patch = json::parse(nested_objects(depth, variant)); + + json result = json::parse(nested_objects(depth, (variant + 1) % 3)); + json expected = result; + result.merge_patch(patch); + reference_merge_patch(expected, patch); + CHECK(result == expected); + + // a target that is not an object, and an empty one + json from_null; + from_null.merge_patch(patch); + json expected_from_null; + reference_merge_patch(expected_from_null, patch); + CHECK(from_null == expected_from_null); + } + } + } + + SECTION("patches nested too deeply for the call stack (#5393)") + { + // applying a patch used to recurse once per nesting level. The result + // is only walked, never copied or compared, since those recurse too. + const std::size_t depth = 100000; + json target = json::parse(nested_objects(depth, 0)); + target.merge_patch(json::parse(nested_objects(depth, 1))); + + const json* p = ⌖ + for (std::size_t i = 0; i < depth; ++i) + { + p = &p->at("a"); + } + // {"y":2} patched with {"x":1,"y":null} + CHECK(p->size() == 1); + CHECK(p->at("x") == 1); + } +} diff --git a/tests/src/unit-modifiers.cpp b/tests/src/unit-modifiers.cpp index 369162772..55f9d467e 100644 --- a/tests/src/unit-modifiers.cpp +++ b/tests/src/unit-modifiers.cpp @@ -11,6 +11,53 @@ #include using nlohmann::json; +#include + +namespace +{ +// update(source, true) as documented, written recursively; only usable on +// values nested a few hundred levels deep +void reference_update(json& target, const json& source) +{ + for (auto it = source.begin(); it != source.end(); ++it) + { + const auto existing = target.find(it.key()); + if (it.value().is_object() && existing != target.end() && existing->is_object()) + { + reference_update(*existing, it.value()); + } + else + { + target[it.key()] = it.value(); + } + } +} + +// objects nested `depth` levels deep under the key "a", with members that +// differ by `variant` on the way down +std::string nested_objects(const std::size_t depth, const int variant) +{ + std::string text; + for (std::size_t i = 0; i < depth; ++i) + { + text += "{"; + if ((i + static_cast(variant)) % 3 == 0) + { + text += "\"s" + std::to_string(variant) + "\":" + std::to_string(i) + ","; + } + if (variant == 2 && i % 5 == 0) + { + // an object replacing a primitive, which is not merged + text += "\"s0\":{\"o\":1},"; + } + text += "\"a\":"; + } + text += variant == 1 ? "{\"x\":1}" : "{\"y\":2}"; + text.append(depth, '}'); + return text; +} +} // namespace + TEST_CASE("modifiers") { SECTION("clear()") @@ -988,3 +1035,44 @@ TEST_CASE("modifiers") } } } + +TEST_CASE("update() on deeply nested values") +{ + SECTION("merging past the descent bound gives the same result") + { + // every depth on either side of where the iterative version takes + // over (detail::recursion_depth_limit(), 128) + for (std::size_t depth = 0; depth <= 300; ++depth) + { + CAPTURE(depth); + for (int variant = 0; variant < 3; ++variant) + { + CAPTURE(variant); + const json source = json::parse(nested_objects(depth, variant)); + json result = json::parse(nested_objects(depth, (variant + 1) % 3)); + json expected = result; + result.update(source, true); + reference_update(expected, source); + CHECK(result == expected); + } + } + } + + SECTION("objects nested too deeply for the call stack (#5545)") + { + // merging used to recurse once per nesting level. The result is only + // walked, never copied or compared, since those recurse too. + const std::size_t depth = 100000; + json target = json::parse(nested_objects(depth, 0)); + target.update(json::parse(nested_objects(depth, 1)), true); + + const json* p = ⌖ + for (std::size_t i = 0; i < depth; ++i) + { + p = &p->at("a"); + } + CHECK(p->size() == 2); + CHECK(p->at("x") == 1); + CHECK(p->at("y") == 2); + } +} From f290b36ad24af1d05c57af8b62111729bb374021 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 25 Sep 2026 08:23:53 +0200 Subject: [PATCH 34/64] Fix CI: use VS 2026 on windows-11-arm and use raw string literals in tests (#5575) The windows-11-arm runner image moved to windows-11-vs2026-arm64, which no longer ships Visual Studio 2022, so the msvc-arm64 job failed at configure time. Use the "Visual Studio 18 2026" generator like the msvc2026 job. clang-tidy's modernize-raw-string-literal check flagged two string literals in the nesting tests added by #5546 and #5547. Signed-off-by: Niels Lohmann --- .github/workflows/windows.yml | 4 ++-- tests/src/unit-merge_patch.cpp | 2 +- tests/src/unit-modifiers.cpp | 2 +- 3 files changed, 4 insertions(+), 4 deletions(-) diff --git a/.github/workflows/windows.yml b/.github/workflows/windows.yml index 99a8aa3ca..ea742773f 100644 --- a/.github/workflows/windows.yml +++ b/.github/workflows/windows.yml @@ -124,11 +124,11 @@ jobs: steps: - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 - name: Run CMake (Release) - run: cmake -S . -B build -G "Visual Studio 17 2022" -A ARM64 -DJSON_BuildTests=On -DCMAKE_CXX_FLAGS="/W4 /WX" + run: cmake -S . -B build -G "Visual Studio 18 2026" -A ARM64 -DJSON_BuildTests=On -DCMAKE_CXX_FLAGS="/W4 /WX" if: matrix.build_type == 'Release' shell: pwsh - name: Run CMake (Debug) - run: cmake -S . -B build -G "Visual Studio 17 2022" -A ARM64 -DJSON_BuildTests=On -DJSON_FastTests=ON -DCMAKE_CXX_FLAGS="/W4 /WX" + run: cmake -S . -B build -G "Visual Studio 18 2026" -A ARM64 -DJSON_BuildTests=On -DJSON_FastTests=ON -DCMAKE_CXX_FLAGS="/W4 /WX" if: matrix.build_type == 'Debug' shell: pwsh - name: Build diff --git a/tests/src/unit-merge_patch.cpp b/tests/src/unit-merge_patch.cpp index 8ac281ef5..c9741e85b 100644 --- a/tests/src/unit-merge_patch.cpp +++ b/tests/src/unit-merge_patch.cpp @@ -62,7 +62,7 @@ std::string nested_objects(const std::size_t depth, const int variant) } text += "\"a\":"; } - text += variant == 1 ? "{\"x\":1,\"y\":null}" : "{\"y\":2}"; + text += variant == 1 ? R"({"x":1,"y":null})" : "{\"y\":2}"; text.append(depth, '}'); return text; } diff --git a/tests/src/unit-modifiers.cpp b/tests/src/unit-modifiers.cpp index 55f9d467e..c878ec15c 100644 --- a/tests/src/unit-modifiers.cpp +++ b/tests/src/unit-modifiers.cpp @@ -48,7 +48,7 @@ std::string nested_objects(const std::size_t depth, const int variant) if (variant == 2 && i % 5 == 0) { // an object replacing a primitive, which is not merged - text += "\"s0\":{\"o\":1},"; + text += R"("s0":{"o":1},)"; } text += "\"a\":"; } From 3901b223e58671d0e8c460b530de2dfe731a269e Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 25 Sep 2026 08:28:00 +0200 Subject: [PATCH 35/64] Complete the architecture documentation page (#5570) * Complete the architecture documentation page Replace the placeholder bullets and TODOs with a description of the component pipeline (with a diagram), the source layout, the template parameters, the value storage (now in struct data), the input and output adapters, and the SAX interface. Signed-off-by: Niels Lohmann * Link sources and basic_json, document full input adapter interface Signed-off-by: Niels Lohmann * Align the default column of the template parameter table Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- docs/mkdocs/docs/home/architecture.md | 212 +++++++++++++++++++++----- 1 file changed, 177 insertions(+), 35 deletions(-) diff --git a/docs/mkdocs/docs/home/architecture.md b/docs/mkdocs/docs/home/architecture.md index aba2be580..6c0892f11 100644 --- a/docs/mkdocs/docs/home/architecture.md +++ b/docs/mkdocs/docs/home/architecture.md @@ -1,34 +1,125 @@ # Architecture -!!! info - - This page is still under construction. Its goal is to provide a high-level overview of the library's architecture. - This should help new contributors to get an idea of the used concepts and where to make changes. +This page gives a high-level overview of the library's architecture. It should help new contributors to get an idea of +the used concepts and where to make changes. ## Overview -The main structure is class [nlohmann::basic_json](../api/basic_json/index.md). +The library is built around a single class template, [`nlohmann::basic_json`](../api/basic_json/index.md). A +`basic_json` value is a node in a tree of JSON values. All other components either create such a tree from an input +(parsing), write a tree to an output (serialization), or give access to it (iterators, JSON Pointer, conversions). -- public API -- container interface -- iterators +```mermaid +flowchart LR + input[/"input
(string, stream,
iterator range, file)"/] + ia["input adapter"] + lexer["lexer"] + parser["parser"] + breader["binary_reader"] + sax["SAX interface"] + value[("basic_json
value tree")] + serializer["serializer"] + bwriter["binary_writer"] + oa["output adapter"] + output[/"output
(string, stream,
vector)"/] -## Template specializations + input --> ia + ia --> lexer --> parser --> sax + ia --> breader --> sax + sax --> value + value --> serializer --> oa + value --> bwriter --> oa + oa --> output +``` -- describe template parameters of `basic_json` -- [`json`](../api/json.md) -- [`ordered_json`](../api/ordered_json.md) via [`ordered_map`](../api/ordered_map.md) +- **JSON text** is read by an [input adapter](#input-adapters), tokenized by the lexer, and turned into SAX events by + the parser. +- **Binary formats** (BJData, BSON, CBOR, MessagePack, UBJSON) are read by an input adapter and turned into the same SAX + events by the `binary_reader`. +- A [SAX consumer](#sax-interface) receives the events. The one used by [`parse`](../api/basic_json/parse.md) builds a + `basic_json` value tree. +- The `serializer` (JSON text) or the `binary_writer` (binary formats) writes a value tree to an + [output adapter](#output-adapters). + +## Source layout + +The public headers are in [`include/nlohmann`](https://github.com/nlohmann/json/tree/develop/include/nlohmann): + +- [`json.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/json.hpp) defines class [`basic_json`](../api/basic_json/index.md). +- [`json_fwd.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/json_fwd.hpp) contains forward declarations. +- [`adl_serializer.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/adl_serializer.hpp), [`byte_container_with_subtype.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/byte_container_with_subtype.hpp), and [`ordered_map.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/ordered_map.hpp) define + [`adl_serializer`](../api/adl_serializer/index.md), + [`byte_container_with_subtype`](../api/byte_container_with_subtype/index.md), and + [`ordered_map`](../api/ordered_map.md). + +Everything else lives in [`detail/`](https://github.com/nlohmann/json/tree/develop/include/nlohmann/detail) and namespace `nlohmann::detail`, which is not part of the public API. Paths +below are relative to `include/nlohmann`. + +| Component | Location | +|-----------|----------| +| Value type enumeration | [`detail/value_t.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/value_t.hpp) | +| Input adapters | [`detail/input/input_adapters.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/input/input_adapters.hpp) | +| Lexer | [`detail/input/lexer.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/input/lexer.hpp), [`detail/input/number_parse.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/input/number_parse.hpp), [`detail/input/string_scan.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/input/string_scan.hpp) | +| Parser | [`detail/input/parser.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/input/parser.hpp) | +| SAX interface and DOM builders | [`detail/input/json_sax.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/input/json_sax.hpp) | +| Binary format readers | [`detail/input/binary_reader.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/input/binary_reader.hpp) | +| JSON serializer | [`detail/output/serializer.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/output/serializer.hpp), [`detail/conversions/to_chars.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/conversions/to_chars.hpp) | +| Binary format writers | [`detail/output/binary_writer.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/output/binary_writer.hpp) | +| Output adapters | [`detail/output/output_adapters.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/output/output_adapters.hpp) | +| Iterators | [`detail/iterators/`](https://github.com/nlohmann/json/tree/develop/include/nlohmann/detail/iterators) | +| Conversions from/to arbitrary types | [`detail/conversions/from_json.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/conversions/from_json.hpp), [`detail/conversions/to_json.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/conversions/to_json.hpp) | +| JSON Pointer | [`detail/json_pointer.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/json_pointer.hpp) | +| Exceptions | [`detail/exceptions.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/exceptions.hpp) | +| Type traits and C++ feature backports | [`detail/meta/`](https://github.com/nlohmann/json/tree/develop/include/nlohmann/detail/meta) | +| Macros | [`detail/macro_scope.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/macro_scope.hpp), [`detail/macro_unscope.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/macro_unscope.hpp), [`detail/abi_macros.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/abi_macros.hpp) | + +The single-header version [`single_include/nlohmann/json.hpp`](https://github.com/nlohmann/json/blob/develop/single_include/nlohmann/json.hpp) +is generated from these files with `make amalgamate` and must not be edited by hand. + +## Template parameters + +[`basic_json`](../api/basic_json/index.md) is parameterized by the types it uses to store values and to convert from and to other types: + +| Template parameter | Default | Used for | +|----------------------|-----------------------------|-------------------------------------------------------------------| +| `ObjectType` | `std::map` | objects, see [`object_t`](../api/basic_json/object_t.md) | +| `ArrayType` | `std::vector` | arrays, see [`array_t`](../api/basic_json/array_t.md) | +| `StringType` | `std::string` | strings and object keys, see [`string_t`](../api/basic_json/string_t.md) | +| `BooleanType` | `bool` | Booleans, see [`boolean_t`](../api/basic_json/boolean_t.md) | +| `NumberIntegerType` | `std::int64_t` | signed integers, see [`number_integer_t`](../api/basic_json/number_integer_t.md) | +| `NumberUnsignedType` | `std::uint64_t` | unsigned integers, see [`number_unsigned_t`](../api/basic_json/number_unsigned_t.md) | +| `NumberFloatType` | `double` | floating-point numbers, see [`number_float_t`](../api/basic_json/number_float_t.md) | +| `AllocatorType` | `std::allocator` | allocating objects, arrays, strings, and binary values | +| `JSONSerializer` | `adl_serializer` | conversions from/to other types, see [`adl_serializer`](../api/adl_serializer/index.md) | +| `BinaryType` | `std::vector` | binary values, see [`binary_t`](../api/basic_json/binary_t.md) | +| `CustomBaseClass` | `void` | an optional base class, see [`json_base_class_t`](../api/basic_json/json_base_class_t.md) | + +The library provides two specializations: + +- [`json`](../api/json.md) uses all default template arguments. +- [`ordered_json`](../api/ordered_json.md) uses [`ordered_map`](../api/ordered_map.md) as `ObjectType` to keep the + insertion order of object keys. + +The requirements on the template arguments are listed in +[Template Parameter Requirements](../features/types/template_parameters.md). ## Value storage -Values are stored as a tagged union of [value_t](../api/basic_json/value_t.md) and json_value. +Each [`basic_json`](../api/basic_json/index.md) value stores its content as a tagged union: an enumeration [`value_t`](../api/basic_json/value_t.md) +names the type of the value, and a union `json_value` holds the value itself. Both are members of the nested struct +`data`, which is the only data member `m_data` of `basic_json`: ```cpp -/// the type of the current element -value_t m_type = value_t::null; +struct data +{ + /// the type of the current element + value_t m_type = value_t::null; -/// the value of the current element -json_value m_value = {}; + /// the value of the current element + json_value m_value = {}; +}; + +data m_data = {}; ``` with @@ -68,42 +159,83 @@ union json_value { }; ``` -## Parsing inputs (deserialization) +Objects, arrays, strings, and binary values are allocated on the heap with `AllocatorType`, and the union only stores a +pointer to them. This keeps a `basic_json` value small: one pointer-sized union and one byte for the type. The class +maintains the invariant that the pointer matching `m_type` is never null; `assert_invariant()` checks it with +[runtime assertions](../features/assertions.md). -Input is read via **input adapters** that abstract a source with a common interface: +## Input adapters + +Input is read via **input adapters** that abstract a source. Every input adapter provides this interface: ```cpp -/// read a single character -std::char_traits::int_type get_character() noexcept; +/// the type of the characters in the input +using char_type = ...; -/// read multiple characters to a destination buffer and -/// returns the number of characters successfully read +/// read a single character; returns std::char_traits::eof() at the end of the input +typename std::char_traits::int_type get_character(); + +/// read up to count * sizeof(T) bytes into dest and return the number of bytes read +/// (used by the binary readers) template std::size_t get_elements(T* dest, std::size_t count = 1); ``` -List examples of input adapters. +The lexer detects two optional extensions at compile time. Only `iterator_input_adapter` provides them, and only for +random-access input of single-byte characters: -## SAX Interface +- `supports_seek`, `get_consumed_count()`, and `copy_consumed_range()` let the lexer reconstruct already consumed input + for error messages instead of copying every character it reads. +- `supports_bulk_scan`, `bulk_data()`, `bulk_remaining()`, and `bulk_skip()` let the lexer scan strings directly in + contiguous memory, several bytes at a time. -TODO +The function `input_adapter` picks the right adapter for the argument passed to `parse`, `accept`, `sax_parse`, or the +`from_*` functions: -## Writing outputs (serialization) +- `iterator_input_adapter` reads from an iterator range, which also covers strings, containers, and pointers. +- `wide_string_input_adapter` reads from ranges of `wchar_t`, `char16_t`, or `char32_t` and converts them to UTF-8. + It cannot be used for binary formats; its `get_elements()` throws. +- `input_stream_adapter` reads from a `std::istream`. +- `file_input_adapter` reads from a `std::FILE*`. + +## SAX interface + +The parser does not build values itself. It reports what it reads as events to a [SAX](../features/parsing/sax_interface.md) +consumer, which implements the interface [`json_sax`](../api/json_sax/index.md): `null`, `boolean`, `number_integer`, +`number_unsigned`, `number_float`, `string`, `binary`, `start_object`, `key`, `end_object`, `start_array`, `end_array`, +and `parse_error`. + +The library comes with two consumers in `detail/input/json_sax.hpp`: + +- `json_sax_dom_parser` builds a [`basic_json`](../api/basic_json/index.md) value tree. [`parse`](../api/basic_json/parse.md) uses it. +- `json_sax_dom_callback_parser` does the same, but calls a [parser callback](../features/parsing/parser_callbacks.md) + for each event, which can skip values. `parse` uses it when a callback is given. + +The `binary_reader` emits the same events for binary formats, so [`sax_parse`](../api/basic_json/sax_parse.md) works +with a user-defined consumer for JSON and for all binary formats alike. + +## Output adapters Output is written via **output adapters**: ```cpp -template void write_character(CharType c); -template void write_characters(const CharType* s, std::size_t length); ``` -List examples of output adapters. +The `serializer` (used by [`dump`](../api/basic_json/dump.md) and [`operator<<`](../api/operator_ltlt.md)) and the +`binary_writer` (used by the `to_*` functions) write to one of these adapters: + +- `output_vector_adapter` appends to a `std::vector`. +- `output_stream_adapter` writes to a `std::ostream`. +- `output_string_adapter` appends to a string. ## Value conversion +Values are converted from and to other types with the `JSONSerializer` template parameter. The default, +[`adl_serializer`](../api/adl_serializer/index.md), calls the free functions + ```cpp template void to_json(basic_json& j, const T& t); @@ -112,13 +244,23 @@ template void from_json(const basic_json& j, T& t); ``` +found by argument-dependent lookup. The library defines them for standard types in `detail/conversions`; users add them +for their own types, see [Arbitrary Type Conversions](../features/arbitrary_types.md). The +[serialization macros](../features/macros.md) generate these functions. + ## Additional features -- JSON Pointers -- Binary formats -- Custom base class -- Conversion macros +- [JSON Pointer](../features/json_pointer.md) (class `json_pointer`) addresses values inside a tree. It is also the + basis of [JSON Patch](../features/json_patch.md). +- [Binary formats](../features/binary_formats/index.md) are read by `binary_reader` and written by `binary_writer`. +- A [custom base class](../api/basic_json/json_base_class_t.md) can add members to every [`basic_json`](../api/basic_json/index.md) value. +- [Serialization macros](../features/macros.md) generate `to_json` and `from_json` functions for user-defined types. ## Details namespace -- C++ feature backports +Namespace `nlohmann::detail` contains all implementation details. It is not part of the public API and may change in any +release. Besides the components above, it contains: + +- type traits to detect the capabilities of user-defined types (`detail/meta/type_traits.hpp`), +- backports of C++14/17 features to C++11 (`detail/meta/cpp_future.hpp`), and +- helpers such as `string_concat` and `string_escape`. From aada27405dca0771eca5f8eaa1d087c4b13c0ff0 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 25 Sep 2026 08:28:16 +0200 Subject: [PATCH 36/64] Add a roadmap page to the documentation (#5571) Describe what the project will and will not do over the next year, and point to issue #3453 for the open question of a 4.0 release. Signed-off-by: Niels Lohmann --- docs/mkdocs/docs/community/index.md | 1 + docs/mkdocs/docs/community/roadmap.md | 43 +++++++++++++++++++++++++++ docs/mkdocs/mkdocs.yml | 1 + 3 files changed, 45 insertions(+) create mode 100644 docs/mkdocs/docs/community/roadmap.md diff --git a/docs/mkdocs/docs/community/index.md b/docs/mkdocs/docs/community/index.md index 50baeab25..c7d0cf079 100644 --- a/docs/mkdocs/docs/community/index.md +++ b/docs/mkdocs/docs/community/index.md @@ -5,4 +5,5 @@ - [Contribution Guidelines](contribution_guidelines.md) - guidelines how to contribute to this project - [Governance](governance.md) - the governance model of this project - [Quality Assurance](quality_assurance.md) - how the quality of this project is assured +- [Roadmap](roadmap.md) - what the project will and will not do - [Security Policy](security_policy.md) - the security policy of the project diff --git a/docs/mkdocs/docs/community/roadmap.md b/docs/mkdocs/docs/community/roadmap.md new file mode 100644 index 000000000..e8c407d3f --- /dev/null +++ b/docs/mkdocs/docs/community/roadmap.md @@ -0,0 +1,43 @@ +# Roadmap + +This page describes what the project intends to do, and what it does not intend to do, over the next year. Concrete +work items are tracked in the [GitHub milestones](https://github.com/nlohmann/json/milestones) and the +[issue tracker](https://github.com/nlohmann/json/issues). + +## What the project will do + +- **Keep the C++11 baseline.** The library will continue to compile with every + [supported C++11 compiler](https://github.com/nlohmann/json/blob/develop/README.md#supported-compilers). Features of + later standards are only used when they are guarded by the `JSON_HAS_CPP_*` macros. +- **Stay conformant to JSON.** The parser and serializer follow [RFC 8259](https://datatracker.ietf.org/doc/html/rfc8259). + Extensions such as [comments](../features/comments.md) or [trailing commas](../features/trailing_commas.md) remain + opt-in. +- **Keep the 3.x public API stable.** Releases follow [semantic versioning](https://semver.org). Changes that would + break existing code are only added behind a feature macro, so users can opt in and test their code before a next + major release. +- **Support a broad range of compilers and platforms.** The [CI](quality_assurance.md) keeps testing old and new + versions of GCC, Clang, MSVC, and other compilers on Linux, macOS, and Windows. +- **Keep the quality assurance up.** Every change keeps the test coverage at 100%, passes the static and dynamic + analysis, and is fuzz-tested by OSS-Fuzz, see [Quality assurance](quality_assurance.md). +- **Harden the library against hostile input.** Handling deeply nested values without exhausting the call stack is + ongoing work. +- **Fix bugs and security issues** reported through the issue tracker and the [security policy](security_policy.md). + +## What the project will not do + +- **Break the public API of version 3.x.** See the + [contribution guidelines](https://github.com/nlohmann/json/blob/develop/.github/CONTRIBUTING.md#break-the-public-api) + for what counts as a breaking change. +- **Require a newer C++ standard than C++11.** +- **Break JSON conformance** or enable non-standard extensions by default. +- **Add dependencies** or require a build step. The library remains header-only, and the single header + `json.hpp` remains a complete distribution. +- **Trade simplicity for speed or memory efficiency.** Performance improvements are welcome, but the library is not + meant to compete with the fastest JSON libraries, see [Design goals](../home/design_goals.md). + +## Version 4.0 + +There is no decision yet on whether or when a version 4.0 with breaking changes will be released. Proposals that need +a major version, for instance stricter type conversions, are collected in issue +[#3453](https://github.com/nlohmann/json/issues/3453). Until then, such changes are only added as opt-in behavior +behind feature macros. diff --git a/docs/mkdocs/mkdocs.yml b/docs/mkdocs/mkdocs.yml index 8f92a838f..9400222b1 100644 --- a/docs/mkdocs/mkdocs.yml +++ b/docs/mkdocs/mkdocs.yml @@ -317,6 +317,7 @@ nav: - community/contribution_guidelines.md - community/quality_assurance.md - community/governance.md + - community/roadmap.md - community/security_policy.md # Extras From 80bf54a5a2294f35b3858b602af0d9b795889b0a Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 25 Sep 2026 08:28:34 +0200 Subject: [PATCH 37/64] Add a security assurance case to the documentation (#5572) Describe the threat model, the trust boundaries, the secure-design argument, and how common weaknesses are countered, with links to the quality assurance page as evidence. Signed-off-by: Niels Lohmann --- docs/mkdocs/docs/community/assurance_case.md | 70 ++++++++++++++++++++ docs/mkdocs/docs/community/index.md | 1 + docs/mkdocs/mkdocs.yml | 1 + 3 files changed, 72 insertions(+) create mode 100644 docs/mkdocs/docs/community/assurance_case.md diff --git a/docs/mkdocs/docs/community/assurance_case.md b/docs/mkdocs/docs/community/assurance_case.md new file mode 100644 index 000000000..87d8d6f14 --- /dev/null +++ b/docs/mkdocs/docs/community/assurance_case.md @@ -0,0 +1,70 @@ +# Assurance case + +This page argues why the library meets its security requirements. It describes the threats the library faces, where the +trust boundaries lie, and how the library's design and the [quality assurance](quality_assurance.md) counter these +threats. To report a vulnerability, see the [security policy](security_policy.md). + +## Threat model + +The library parses, stores, and serializes JSON values in memory. It does not open network connections, does not open +files (it only reads from streams or `std::FILE*` handles that the caller has already opened), does not read environment +variables, and does not implement cryptography or handle credentials. + +The primary threat is therefore **untrusted input**: JSON text or binary data (BJData, BSON, CBOR, MessagePack, UBJSON) +that an attacker controls, passed to [`parse`](../api/basic_json/parse.md), [`accept`](../api/basic_json/accept.md), +[`sax_parse`](../api/basic_json/sax_parse.md), or one of the `from_*` functions such as +[`from_cbor`](../api/basic_json/from_cbor.md). Such input may try to + +- make the library read or write out of bounds (malformed lengths, truncated input, invalid UTF-8), +- trigger undefined behavior (integer overflow in sizes or numbers, invalid casts), +- exhaust memory (huge announced sizes), or +- exhaust the call stack (deeply nested arrays and objects). + +## Trust boundaries + +- **Untrusted:** all serialized input read by the parser, the SAX interface, and the binary readers. The library must + handle every possible input by either producing a value or throwing a [`parse_error`](../home/exceptions.md#parse-errors) + (or returning `false` when exceptions are disabled for the call). +- **Trusted:** the C++ code that calls the library. Calling a function with violated preconditions, for instance + accessing an array with [`operator[]`](../api/basic_json/operator%5B%5D.md) out of range, is a programming error and + not a security boundary. Such preconditions are checked with [runtime assertions](../features/assertions.md) in debug + builds; functions such as [`at`](../api/basic_json/at.md) offer checked access with exceptions. + +## Secure design + +- **Strict parsing.** The parser accepts exactly the JSON grammar of [RFC 8259](https://datatracker.ietf.org/doc/html/rfc8259). + Extensions such as [comments](../features/comments.md) and [trailing commas](../features/trailing_commas.md) must be + enabled explicitly. Invalid UTF-8 is rejected. +- **Errors are reported, not ignored.** Malformed input results in a [`parse_error`](../home/exceptions.md#parse-errors) + with the byte position of the error. Binary readers do not trust announced sizes: strings and binary values grow + only as bytes are actually read, arrays reserve at most a fixed number of elements up front, and sizes that no + container can hold are rejected. +- **Memory is owned by values.** Each `basic_json` value owns its content, and there is no manual memory management in + user code. The destructor does not recurse, so destroying a deeply nested value does not exhaust the stack. +- **Bounded recursion.** The JSON parser and the binary readers keep their state in explicit stacks instead of + recursing per nesting level. Operations that walk a value, such as [`dump`](../api/basic_json/dump.md), copying, + hashing, and [`merge_patch`](../api/basic_json/merge_patch.md), recurse only up to a fixed depth and continue with an + explicit stack below it. Some operations, such as comparison, [`diff`](../api/basic_json/diff.md), + [`flatten`](../api/basic_json/flatten.md), and the binary writers, still recurse once per nesting level; work on them + is in progress. Applications that process untrusted input can limit its nesting depth with a + [parser callback](../features/parsing/parser_callbacks.md). +- **Invariants are checked.** The class invariant (for instance, that the pointer for the stored type is never null) is + checked with runtime assertions throughout the test suite. + +## Common weaknesses + +The following table maps the relevant classes of the [Common Weakness Enumeration](https://cwe.mitre.org) to the +measures that counter them. The measures are described in detail in [Quality assurance](quality_assurance.md). + +| Weakness | Countermeasures | +|---------------------------------------------------------------------------|--------------------------------------------------------------------------------------------------| +| Out-of-bounds read/write ([CWE-125](https://cwe.mitre.org/data/definitions/125.html), [CWE-787](https://cwe.mitre.org/data/definitions/787.html)) | bounds checks on all reads from the input; AddressSanitizer and Valgrind on the test suite; OSS-Fuzz | +| Integer overflow ([CWE-190](https://cwe.mitre.org/data/definitions/190.html)) | UndefinedBehaviorSanitizer with integer overflow detection; Clang-Tidy; Cppcheck | +| Use after free, double free ([CWE-416](https://cwe.mitre.org/data/definitions/416.html), [CWE-415](https://cwe.mitre.org/data/definitions/415.html)) | ownership of all memory by values; AddressSanitizer and Valgrind; Clang Static Analyzer | +| Memory leaks ([CWE-401](https://cwe.mitre.org/data/definitions/401.html)) | Valgrind (Memcheck) on the test suite | +| Uncontrolled recursion ([CWE-674](https://cwe.mitre.org/data/definitions/674.html)) | iterative parser, binary readers, and destructor; bounded recursion in value operations; tests with deeply nested inputs | +| Uncontrolled resource consumption ([CWE-400](https://cwe.mitre.org/data/definitions/400.html)) | allocations based on announced sizes are capped; OSS-Fuzz with memory limits | +| Undefined behavior in general ([CWE-758](https://cwe.mitre.org/data/definitions/758.html)) | UndefinedBehaviorSanitizer; runtime assertions; Clang-Tidy, Cppcheck, Clang Static Analyzer, Infer | + +In addition, every line of the library is covered by the unit tests, and all parsers are fuzz-tested around the clock +by [OSS-Fuzz](https://github.com/google/oss-fuzz/tree/master/projects/json). diff --git a/docs/mkdocs/docs/community/index.md b/docs/mkdocs/docs/community/index.md index c7d0cf079..7b7f5c07a 100644 --- a/docs/mkdocs/docs/community/index.md +++ b/docs/mkdocs/docs/community/index.md @@ -7,3 +7,4 @@ - [Quality Assurance](quality_assurance.md) - how the quality of this project is assured - [Roadmap](roadmap.md) - what the project will and will not do - [Security Policy](security_policy.md) - the security policy of the project +- [Assurance Case](assurance_case.md) - why the library meets its security requirements diff --git a/docs/mkdocs/mkdocs.yml b/docs/mkdocs/mkdocs.yml index 9400222b1..ffb0fae80 100644 --- a/docs/mkdocs/mkdocs.yml +++ b/docs/mkdocs/mkdocs.yml @@ -319,6 +319,7 @@ nav: - community/governance.md - community/roadmap.md - community/security_policy.md + - community/assurance_case.md # Extras extra: From cc472af13f8a09d4caccc9df755435d06efaa3a9 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 25 Sep 2026 08:29:02 +0200 Subject: [PATCH 38/64] Check the fuzzers' UBJSON/BJData round-trip invariants in the unit tests (#5569) * Check the fuzzers' UBJSON/BJData round-trip invariants in the unit tests The strongest correctness checks for the UBJSON and BJData writers lived only in the OSS-Fuzz drivers: anything from_ubjson()/from_bjdata() returns must serialize with every option combination, parse back, and re-serialize stably. Those checks only run at OSS-Fuzz, so regressions surfaced days later as external reports - the same BJData assert pair was reported five times over three years, and #5494's harness change was followed by OSS-Fuzz 563659413 within a day. Add "UBJSON round-trip invariants" and "BJData round-trip invariants" test cases that run the drivers' checks on a fixed, deterministic corpus (tests/src/round_trip_corpus.hpp): integer and float boundaries, non-finite numbers, strings, binary values, optimized containers, deep nesting, the JData annotated-array matrix, and seeded random containers. They also check two properties the drivers do not: the first round trip preserves the value, and re-serializing reproduces the exact bytes. For BJData both exclude values containing a binary value, which is read back as an array of integers unless it was written as a Draft 3 optimized binary array; this carve-out is now documented in bjdata.md. Run against the headers before #5542, the BJData test fails, including on the shape from OSS-Fuzz 563659413. Also document how OSS-Fuzz reports are handled (reference them as "OSS-Fuzz: ", turn the reproducer into a unit test, keep drivers and unit tests in sync) in tests/fuzzing.md, and link it from the PR template and the quality assurance page. Signed-off-by: Niels Lohmann * Add the OSS-Fuzz reproducers for 474400817 and 474480402 as unit tests Following the convention added to tests/fuzzing.md, the reproducers of the two BJData fuzzer asserts tracked since January are now unit tests: - 474400817 (assert(false)): an empty object _ArraySize_ was written as the ND-array header length, which from_bjdata() could not read back. Fixed by #5455. - 474480402 (to_bjdata(j2, false, false) == vec2): a one-byte Draft 3 binary array is written in Draft 2 mode as a uint8 array and then re-serialized with the int8 marker. This is the documented exception to byte stability, not a library bug; OSS-Fuzz closed it after #5494 relaxed the harness to value stability. The test pins the exact bytes so the exception stays deliberate. The 563659413 reproducer is already a unit test (#5542). A comment also ties the existing UBJSON excessive-count test to the timeout OSS-Fuzz reported for that shape (testcase 6347769435193344). OSS-Fuzz: 474400817 OSS-Fuzz: 474480402 Signed-off-by: Niels Lohmann * Fix GCC -Weffc++ and -Wuseless-cast warnings in the round-trip corpus Initialize the atoms in the member initialization list, and drop the cast of the generator's result, which already is std::size_t on 64-bit Linux. Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- .github/PULL_REQUEST_TEMPLATE.md | 1 + .../docs/community/quality_assurance.md | 3 + .../docs/features/binary_formats/bjdata.md | 10 + tests/fuzzing.md | 23 ++ tests/src/fuzzer-parse_bjdata.cpp | 3 + tests/src/fuzzer-parse_ubjson.cpp | 3 + tests/src/round_trip_corpus.hpp | 213 ++++++++++++++++++ tests/src/unit-bjdata.cpp | 103 +++++++++ tests/src/unit-ubjson.cpp | 50 +++- 9 files changed, 408 insertions(+), 1 deletion(-) create mode 100644 tests/src/round_trip_corpus.hpp diff --git a/.github/PULL_REQUEST_TEMPLATE.md b/.github/PULL_REQUEST_TEMPLATE.md index 537095324..16d808485 100644 --- a/.github/PULL_REQUEST_TEMPLATE.md +++ b/.github/PULL_REQUEST_TEMPLATE.md @@ -2,6 +2,7 @@ - [ ] The changes are described in detail, both the what and why. - [ ] If applicable, an [existing issue](https://github.com/nlohmann/json/issues) is referenced. +- [ ] If applicable, a fixed [OSS-Fuzz](https://issues.oss-fuzz.com) issue is referenced as `OSS-Fuzz: ` (see [fuzz testing](https://github.com/nlohmann/json/blob/develop/tests/fuzzing.md#handling-oss-fuzz-reports)). - [ ] The [Code coverage](https://coveralls.io/github/nlohmann/json) remained at 100%. A test case for every new line of code. - [ ] If applicable, the [documentation](https://json.nlohmann.me) is updated. - [ ] The source code is amalgamated by running `make amalgamate`. diff --git a/docs/mkdocs/docs/community/quality_assurance.md b/docs/mkdocs/docs/community/quality_assurance.md index 4196f3532..bd35516b8 100644 --- a/docs/mkdocs/docs/community/quality_assurance.md +++ b/docs/mkdocs/docs/community/quality_assurance.md @@ -164,6 +164,9 @@ Note: Some modern features (like C++20 ranges or filesystem support) may be disa - [x] The parser is tested against extensive correctness suites for JSON compliance. - [x] In addition, the library is continuously fuzz-tested at [OSS-Fuzz](https://google.github.io/oss-fuzz/) where the library is checked against billions of inputs. +- [x] Every crash reported by OSS-Fuzz is fixed together with a unit test that reproduces it, and the fix references + the OSS-Fuzz issue. The round-trip checks of the fuzzer drivers are also part of the unit tests. See the + [fuzz testing documentation](https://github.com/nlohmann/json/blob/develop/tests/fuzzing.md#handling-oss-fuzz-reports). ## Static analysis diff --git a/docs/mkdocs/docs/features/binary_formats/bjdata.md b/docs/mkdocs/docs/features/binary_formats/bjdata.md index 4cb61053f..a0c84edaf 100644 --- a/docs/mkdocs/docs/features/binary_formats/bjdata.md +++ b/docs/mkdocs/docs/features/binary_formats/bjdata.md @@ -208,6 +208,16 @@ The library maps BJData types to JSON value types as follows: The mapping is **complete** in the sense that any BJData value can be converted to a JSON value. +!!! info "Round trips" + + A value returned by [`from_bjdata`](../../api/basic_json/from_bjdata.md) can be serialized with + [`to_bjdata`](../../api/basic_json/to_bjdata.md) using any combination of options and parsed back into an equal + value, and serializing that value again with the same options produces the same bytes. The exception is binary + values: they are only written as an optimized binary array (`[$B`) if Draft 3 is enabled and both `use_size` and + `use_type` are set. Otherwise, they are written as arrays of integers and parsed back as such (see the notes on + binary values above), and serializing such an array again may choose different, but equally valid, type markers. + The bytes can then differ, but parsing them again yields the same value. + ??? example ```cpp diff --git a/tests/fuzzing.md b/tests/fuzzing.md index cfbf4f249..b3bf90d5d 100644 --- a/tests/fuzzing.md +++ b/tests/fuzzing.md @@ -79,3 +79,26 @@ the same `fuzzers` target as above and also relies on the `FUZZER_ENGINE` variab [build script](https://github.com/google/oss-fuzz/blob/master/projects/json/build.sh) for more information. In case the build at OSS-Fuzz fails, an issue will be created automatically. + +### Handling OSS-Fuzz reports + +OSS-Fuzz files the crashes it finds in its own [issue tracker](https://issues.oss-fuzz.com), not on GitHub. So that +each report can be traced to the change that fixed it, and each fix to the report it answers, fixes follow these +conventions: + +- **Reference the OSS-Fuzz issue in the pull request**, next to any GitHub issue it closes, as `OSS-Fuzz: ` (for + example, `OSS-Fuzz: 563659413`), and in the commit message. The ID alone does not disclose the crash. If the report + was triaged into a GitHub issue, link the OSS-Fuzz issue there too. +- **Turn the reproducer into a unit test.** Download the testcase from the OSS-Fuzz report, reduce it if possible, and + add it as a regression test to the unit test of the affected format (e.g., `tests/src/unit-bjdata.cpp`), with a + comment naming the OSS-Fuzz issue. This way the input is checked by every CI run rather than only by OSS-Fuzz, and + it stays covered even if OSS-Fuzz later closes the report as not reproducible. +- **Keep the fuzzer drivers and the unit tests in sync.** The round-trip checks of the UBJSON and BJData drivers are + also run on a fixed corpus in the unit tests (see `tests/src/round_trip_corpus.hpp` and the "round-trip invariants" + test cases), so a regression shows up in CI first. When a driver's checks change, change the unit tests with them. +- **Record in the report whether the bug shipped.** OSS-Fuzz asks whether a crash was a short-lived regression or + affects a released version; answer it when the fix is merged, as it decides whether the fix needs a release note or + a security advisory (see the [security policy](../.github/SECURITY.md)). + +After the fix is merged, OSS-Fuzz re-runs the reproducer on its next build and marks the report as verified and +closed. If it does not, the fix is incomplete. diff --git a/tests/src/fuzzer-parse_bjdata.cpp b/tests/src/fuzzer-parse_bjdata.cpp index 41c51a311..d3c9e7a33 100644 --- a/tests/src/fuzzer-parse_bjdata.cpp +++ b/tests/src/fuzzer-parse_bjdata.cpp @@ -42,6 +42,9 @@ dump() serializes any non-finite double the same deterministic way (as JSON `null`, since JSON itself cannot represent NaN/Infinity), so comparing dumps is stable under exactly the same values that break operator==. +The unit tests run the same checks on a fixed corpus (see the "BJData round-trip +invariants" test case), so keep both in sync. + The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer drivers. */ diff --git a/tests/src/fuzzer-parse_ubjson.cpp b/tests/src/fuzzer-parse_ubjson.cpp index 20c20eda7..ebf775b59 100644 --- a/tests/src/fuzzer-parse_ubjson.cpp +++ b/tests/src/fuzzer-parse_ubjson.cpp @@ -21,6 +21,9 @@ array data, it performs the following steps: - j4 = from_ubjson(vec3) - assert(j1 == j4) +The unit tests run the same checks on a fixed corpus (see the "UBJSON round-trip +invariants" test case), so keep both in sync. + The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer drivers. */ diff --git a/tests/src/round_trip_corpus.hpp b/tests/src/round_trip_corpus.hpp new file mode 100644 index 000000000..41cdac3e6 --- /dev/null +++ b/tests/src/round_trip_corpus.hpp @@ -0,0 +1,213 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +#include // nan +#include // size_t +#include // int32_t, int64_t, uint32_t, uint64_t +#include // numeric_limits +#include // mt19937 +#include // string, to_string +#include // move +#include // vector + +#include + +// Values for the round-trip property tests of the UBJSON and BJData writers. +// +// The fuzzer drivers (tests/src/fuzzer-parse_ubjson.cpp and +// fuzzer-parse_bjdata.cpp) check that anything the library parses can be +// serialized, parsed back, and serialized again without loss. Those checks +// only run at OSS-Fuzz, so a regression used to surface days later as an +// external report. The unit tests run the same checks on this corpus in CI. +// +// The corpus is deterministic: std::mt19937's output sequence is fixed by +// the standard, and it is used directly rather than through a distribution +// (whose results are implementation-defined). +namespace utils +{ + +class round_trip_corpus +{ + public: + using json = nlohmann::json; + + static std::vector values() + { + round_trip_corpus corpus; + return corpus.build(); + } + + // whether a value contains a binary value, which a BJData or UBJSON round + // trip may turn into an array of integers + static bool contains_binary(const json& j) + { + if (j.is_binary()) + { + return true; + } + if (j.is_structured()) + { + for (const auto& element : j) + { + if (contains_binary(element)) + { + return true; + } + } + } + return false; + } + + private: + std::vector atoms; + // a fixed seed is the point: the corpus must be the same in every run + std::mt19937 generator{42}; // NOLINT(cert-msc32-c,cert-msc51-cpp,bugprone-random-generator-seed) + + round_trip_corpus() + : atoms + { + nullptr, true, false, + // integers at the boundaries of every UBJSON/BJData integer type + 0, 1, -1, 127, 128, 255, 256, -128, -129, + 32767, 32768, 65535, 65536, -32768, -32769, + (std::numeric_limits::min)(), (std::numeric_limits::max)(), + (std::numeric_limits::max)(), + (std::numeric_limits::min)(), (std::numeric_limits::max)(), + static_cast((std::numeric_limits::max)()) + 1u, + (std::numeric_limits::max)(), + // floating-point numbers, including non-finite ones + 0.0, -0.0, 1.5, -2.25, 3.4e38, (std::numeric_limits::max)(), + std::nan(""), std::numeric_limits::infinity(), -std::numeric_limits::infinity(), + // strings, including a non-ASCII one and one longer than 255 bytes + "", "a", "\xC3\xA4", std::string(300, 'x'), + // binary values with and without subtype + json::binary({}), json::binary({1, 2, 255}), json::binary({0x80, 0x7F}, 42), json::binary({1}, 0) + } + {} + + std::vector build() + { + std::vector result = atoms; + + // each atom inside containers, including homogeneous ones that the + // writers encode as optimized (typed) containers + result.emplace_back(json::array()); + result.emplace_back(json::object()); + for (const auto& atom : atoms) + { + result.push_back(json::array({atom})); + result.push_back(json::array({atom, atom, atom})); + result.push_back(json::array({json::array({atom})})); + result.push_back(json::object({{"key", atom}})); + } + result.push_back(json::array({1, 1.5})); + result.push_back(json::array({-1, 255})); + result.push_back(json::array({"a", "b"})); + + // deep, but well below any recursion or depth limit + json nested_array = 1; + json nested_object = 1; + for (int i = 0; i < 300; ++i) + { + nested_array = json::array({nested_array}); + nested_object = json::object({{"key", nested_object}}); + } + result.push_back(nested_array); + result.push_back(nested_object); + + add_annotated_arrays(result); + add_random_values(result); + return result; + } + + // objects in the JData annotated array format, which the BJData writer + // encodes as ND-arrays when the annotation describes a packed array, and + // as plain objects otherwise (see #5398, #5399, #5403, #5404, and #5542) + static void add_annotated_arrays(std::vector& result) + { + const std::vector types = + { + "uint8", "int8", "uint16", "int16", "uint32", "int32", "uint64", "int64", + "single", "double", "char", "byte", "bool", "unknown", 5, nullptr + }; + const std::vector sizes = + { + json::array(), {3}, {1, 3}, {3, 1}, {2, 3}, {2, 0}, {0, 2}, {2, 2, 2}, {-1, 2}, {2, 1.5}, + "3", 3, nullptr, json::binary({}) + }; + const std::vector data = + { + nullptr, 5, "s", json::object({{"a", 1}}), json::array(), + {1, 2, 3}, {1, 2, 3, 4, 5, 6}, {1, 2, 3, 4, 5, 6, 7, 8}, + {1.5, 2.5, 3.5, 4.5, 5.5, 6.5}, {300, -300, 70000, -70000, 1, 2}, + {"a", "b", "c", "d", "e", "f"}, {json::array({1, 2, 3}), json::array({4, 5, 6})} + }; + + for (const auto& type : types) + { + for (const auto& size : sizes) + { + for (const auto& d : data) + { + result.push_back({{"_ArrayType_", type}, {"_ArraySize_", size}, {"_ArrayData_", d}}); + } + } + } + + // incomplete annotations and annotations with an extra key + result.push_back({{"_ArraySize_", {2, 3}}, {"_ArrayData_", {1, 2, 3, 4, 5, 6}}}); + result.push_back({{"_ArrayType_", "uint8"}, {"_ArrayData_", {1, 2, 3, 4, 5, 6}}}); + result.push_back({{"_ArrayType_", "uint8"}, {"_ArraySize_", {2, 3}}}); + result.push_back({{"_ArrayType_", "uint8"}, {"_ArraySize_", {2, 3}}, {"_ArrayData_", {1, 2, 3, 4, 5, 6}}, {"extra", 1}}); + } + + // random containers of atoms, both homogeneous and mixed + void add_random_values(std::vector& result) + { + for (int i = 0; i < 1000; ++i) + { + result.push_back(random_value(0)); + } + } + + std::size_t random_below(std::size_t bound) + { + return generator() % bound; + } + + json random_value(int depth) + { + const auto kind = random_below(10); + if (depth > 3 || kind < 5) + { + return atoms[random_below(atoms.size())]; + } + + json result = kind < 8 ? json::array() : json::object(); + const auto count = random_below(5); + const bool homogeneous = random_below(2) == 0; + const json fixed = atoms[random_below(atoms.size())]; + for (std::size_t i = 0; i < count; ++i) + { + json element = homogeneous ? fixed : random_value(depth + 1); + if (result.is_array()) + { + result.push_back(std::move(element)); + } + else + { + result[std::to_string(i)] = std::move(element); + } + } + return result; + } +}; + +} // namespace utils diff --git a/tests/src/unit-bjdata.cpp b/tests/src/unit-bjdata.cpp index 9bba141d2..ebbbbfaf6 100644 --- a/tests/src/unit-bjdata.cpp +++ b/tests/src/unit-bjdata.cpp @@ -19,6 +19,7 @@ using nlohmann::json; #include #include #include "make_test_data_available.hpp" +#include "round_trip_corpus.hpp" #include "test_utils.hpp" namespace @@ -2867,6 +2868,21 @@ TEST_CASE("BJData") const auto out_num = json::to_bjdata(j_num); CHECK(out_num.at(0) == '{'); CHECK(json::from_bjdata(out_num) == j_num); + + // OSS-Fuzz issue 474400817: an empty object _ArraySize_ was + // written as the ND-array header length, which from_bjdata() + // could not read back + const std::vector input = + { + '[', '{', 'U', 11, '_', 'A', 'r', 'r', 'a', 'y', 'D', 'a', 't', 'a', '_', 'Z', + 'U', 11, '_', 'A', 'r', 'r', 'a', 'y', 'T', 'y', 'p', 'e', '_', 'S', 'i', 5, 'i', 'n', 't', '1', '6', + 'U', 11, '_', 'A', 'r', 'r', 'a', 'y', 'S', 'i', 'z', 'e', '_', '{', '}', '}', ']' + }; + const json j1 = json::from_bjdata(input); + CHECK(j1 == json::parse(R"([{"_ArrayType_":"int16","_ArraySize_":{},"_ArrayData_":null}])")); + json j2; + CHECK_NOTHROW(j2 = json::from_bjdata(json::to_bjdata(j1, false, false))); + CHECK(j2 == j1); } SECTION("ndarray with out-of-range _ArrayData_ elements stays as object") @@ -4273,6 +4289,93 @@ TEST_CASE("BJData use_type requires use_size") } } +TEST_CASE("BJData round-trip invariants") +{ + // This checks what the parse_bjdata_fuzzer driver checks (see + // tests/src/fuzzer-parse_bjdata.cpp), so that a regression shows up in CI + // rather than as an OSS-Fuzz report: every value from_bjdata() returns + // (j1) can be serialized with any combination of options, the result can + // be parsed back (j2), and serializing j2 again with the same options + // yields a value-equal result. + // + // Beyond the driver, this also checks that j2 equals j1 and that + // serializing j2 reproduces the exact bytes, both except for values that + // contain a binary value: a binary value is only written as a binary + // value with Draft 3's optimized binary array, and otherwise read back as + // an array of integers, for which the writer may choose different (but + // equally valid) type markers when it is serialized again (see #5494). + // + // Values are compared with dump() rather than operator==, because a NaN + // never compares equal to itself. + struct options + { + bool use_size; + bool use_type; + json::bjdata_version_t version; + }; + const std::vector all_options = + { + {false, false, json::bjdata_version_t::draft2}, + {true, false, json::bjdata_version_t::draft2}, + {true, true, json::bjdata_version_t::draft2}, + {false, false, json::bjdata_version_t::draft3}, + {true, false, json::bjdata_version_t::draft3}, + {true, true, json::bjdata_version_t::draft3}, + }; + + for (const auto& j0 : utils::round_trip_corpus::values()) + { + // turn the corpus value into a value as from_bjdata() returns it + for (const auto& initial : all_options) + { + const json j1 = json::from_bjdata(json::to_bjdata(j0, initial.use_size, initial.use_type, initial.version)); + const bool has_binary = utils::round_trip_corpus::contains_binary(j1); + + for (const auto& o : all_options) + { + INFO("j1 = " << j1.dump() << ", use_size = " << o.use_size << ", use_type = " << o.use_type + << ", draft3 = " << (o.version == json::bjdata_version_t::draft3)); + + const std::vector vec = json::to_bjdata(j1, o.use_size, o.use_type, o.version); + json j2; + // anything the library writes must be parsable by the library + REQUIRE_NOTHROW(j2 = json::from_bjdata(vec)); + const std::vector vec2 = json::to_bjdata(j2, o.use_size, o.use_type, o.version); + CHECK(json::from_bjdata(vec2).dump() == j2.dump()); + + if (!has_binary) + { + CHECK(j2.dump() == j1.dump()); + CHECK(vec2 == vec); + } + } + } + } +} + +TEST_CASE("BJData round trip of a binary value is value-stable, not byte-stable") +{ + // OSS-Fuzz issue 474480402: a Draft 3 optimized binary array is read as a + // binary value, which to_bjdata() writes in the default Draft 2 mode as a + // plain array of uint8 numbers. That is read back as an array of numbers, + // for which the writer then picks the smallest type marker, int8 ('i'), + // so re-serializing changes the bytes, but not the value. This is the + // exception described in the "Round trips" note of the BJData + // documentation, and why the fuzzer checks value stability (see #5494). + const std::vector input = {'[', '$', 'B', '#', 'U', 1, 0x20}; + const json j1 = json::from_bjdata(input); + CHECK(j1 == json::binary({0x20})); + + const std::vector vec = json::to_bjdata(j1, false, false); + CHECK(vec == std::vector({'[', 'U', 0x20, ']'})); + const json j2 = json::from_bjdata(vec); + CHECK(j2 == json::array({0x20})); + + const std::vector vec2 = json::to_bjdata(j2, false, false); + CHECK(vec2 == std::vector({'[', 'i', 0x20, ']'})); + CHECK(json::from_bjdata(vec2) == j2); +} + TEST_CASE("BJData roundtrips" * doctest::skip()) { SECTION("input from self-generated BJData files") diff --git a/tests/src/unit-ubjson.cpp b/tests/src/unit-ubjson.cpp index aafbbf5a4..c8458c44d 100644 --- a/tests/src/unit-ubjson.cpp +++ b/tests/src/unit-ubjson.cpp @@ -15,6 +15,7 @@ using nlohmann::json; #include #include #include "make_test_data_available.hpp" +#include "round_trip_corpus.hpp" #include "test_utils.hpp" namespace @@ -2265,7 +2266,9 @@ TEST_CASE("UBJSON optimized arrays of a valueless type are bounded") SECTION("an excessive count is rejected") { - // 'l' is a big-endian int32: 0x7FFFFFFF elements, about 34 GB of value + // 'l' is a big-endian int32: 0x7FFFFFFF elements, about 34 GB of value; + // OSS-Fuzz reported this shape as a parse_ubjson_fuzzer timeout + // (testcase 6347769435193344, no issue filed) for (const auto marker : {'Z', 'T', 'F' }) @@ -2817,6 +2820,51 @@ TEST_CASE("UBJSON use_type requires use_size") } } +TEST_CASE("UBJSON round-trip invariants") +{ + // This checks what the parse_ubjson_fuzzer driver checks (see + // tests/src/fuzzer-parse_ubjson.cpp), so that a regression shows up in CI + // rather than as an OSS-Fuzz report: every value from_ubjson() returns + // (j1) can be serialized with any combination of options, the result can + // be parsed back (j2), and serializing j2 again with the same options + // reproduces the exact bytes. Beyond the driver, this also checks that j2 + // equals j1. Values are compared with dump() rather than operator==, + // because a NaN never compares equal to itself. + struct options + { + bool use_size; + bool use_type; + }; + const std::vector all_options = + { + {false, false}, + {true, false}, + {true, true}, + }; + + for (const auto& j0 : utils::round_trip_corpus::values()) + { + // turn the corpus value into a value as from_ubjson() returns it; this + // has no binary values, as UBJSON writes them as arrays of integers + for (const auto& initial : all_options) + { + const json j1 = json::from_ubjson(json::to_ubjson(j0, initial.use_size, initial.use_type)); + + for (const auto& o : all_options) + { + INFO("j1 = " << j1.dump() << ", use_size = " << o.use_size << ", use_type = " << o.use_type); + + const std::vector vec = json::to_ubjson(j1, o.use_size, o.use_type); + json j2; + // anything the library writes must be parsable by the library + REQUIRE_NOTHROW(j2 = json::from_ubjson(vec)); + CHECK(j2.dump() == j1.dump()); + CHECK(json::to_ubjson(j2, o.use_size, o.use_type) == vec); + } + } + } +} + TEST_CASE("UBJSON roundtrips" * doctest::skip()) { SECTION("input from self-generated UBJSON files") From 01b53c8c15ae94e3b790ec9578a43966f0262c30 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 25 Sep 2026 08:29:36 +0200 Subject: [PATCH 39/64] Keep JSON_DIAGNOSTICS parent pointers of ordered_json members after erase() and update() (#5552) * Keep JSON_DIAGNOSTICS parent pointers of ordered_json members after erase() and update() ordered_json stores its members in a vector, and two operations moved members without restoring their parent pointers afterwards: - ordered_map::erase() re-constructs every member after the erased one in place. The basic_json move constructor leaves m_parent at nullptr, and none of the object branches of basic_json::erase() (by key, iterator, or iterator range) called set_parents(). This also affected merge_patch() with a null member and patch() with a remove operation. - update() only set the parent pointer of the inserted member. Adding a key can reallocate the vector, which copies all other members and leaves their m_parent at nullptr. The set_parents() call added for #4813 only repaired this for the nested object of a merge, not for the target. The next assert_invariant() on such an object (for instance, when copying it) aborted, and diagnostic messages lost the path prefix above the moved member. std::map-based json was not affected, because its nodes do not move. Erasing from an ordered_map object now calls set_parents(), and update() uses set_parent(), which already refreshes all members for vector-based objects. This makes the #4813 workaround redundant. Signed-off-by: Niels Lohmann * Account for JSON_DIAGNOSTIC_POSITIONS in the ordered_json parent-pointer test The merge_patch() case parses its input, so with JSON_DIAGNOSTIC_POSITIONS the exception message also carries the byte range of the parsed value. Signed-off-by: Niels Lohmann * Silence clang-tidy for the intentional copy in the ordered_json parent-pointer test Signed-off-by: Niels Lohmann * Keep parent pointers when update() merges past its descent bound The iterative path of update() only set the parent pointer of the member it inserted, like the recursive one did before. It now uses set_parent() too, so ordered_json members that move when a nested object grows keep their parents, and the set_parents() calls that patched this up after each nested merge are gone. Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- include/nlohmann/json.hpp | 48 ++++++++++----- single_include/nlohmann/json.hpp | 48 ++++++++++----- tests/src/unit-diagnostics.cpp | 100 +++++++++++++++++++++++++++++++ 3 files changed, 166 insertions(+), 30 deletions(-) diff --git a/include/nlohmann/json.hpp b/include/nlohmann/json.hpp index e40a2aa27..09dca4694 100644 --- a/include/nlohmann/json.hpp +++ b/include/nlohmann/json.hpp @@ -1243,6 +1243,27 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } + /// @brief restore the parent pointers after erasing from an object + /// ordered_json keeps its members in a vector, and erasing a member + /// re-constructs every member after it in place, which resets their + /// parent pointers + void set_parents_after_object_erase() + { +#if JSON_DIAGNOSTICS +#ifdef JSON_HEDLEY_MSVC_VERSION +#pragma warning(push ) +#pragma warning(disable : 4127) // ignore warning to replace if with if constexpr +#endif + if (detail::is_ordered_map::value) + { + set_parents(); + } +#ifdef JSON_HEDLEY_MSVC_VERSION +#pragma warning( pop ) +#endif +#endif + } + public: ////////////////////////// // JSON parser callback // @@ -2931,6 +2952,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec case value_t::object: { result.m_it.object_iterator = erase_from_object(pos.m_it.object_iterator); + set_parents_after_object_erase(); break; } @@ -3003,6 +3025,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec { result.m_it.object_iterator = m_data.m_value.object->erase(first.m_it.object_iterator, last.m_it.object_iterator); + set_parents_after_object_erase(); break; } @@ -3033,7 +3056,9 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_THROW(type_error::create(307, detail::concat("cannot use erase() with ", type_name()), this)); } - return m_data.m_value.object->erase(std::forward(key)); + const auto erased = m_data.m_value.object->erase(std::forward(key)); + set_parents_after_object_erase(); + return erased; } template < typename KeyType, detail::enable_if_t < @@ -3050,6 +3075,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec if (it != m_data.m_value.object->end()) { m_data.m_value.object->erase(it); + set_parents_after_object_erase(); return 1; } return 0; @@ -3961,16 +3987,12 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec if (it2 != m_data.m_value.object->end() && it2->second.is_object()) { it2->second.update_members(it.value().cbegin(), it.value().cend(), true, depth + 1); -#if JSON_DIAGNOSTICS - it2->second.set_parents(); -#endif continue; } } - m_data.m_value.object->operator[](it.key()) = it.value(); -#if JSON_DIAGNOSTICS - m_data.m_value.object->operator[](it.key()).m_parent = this; -#endif + // set_parent() also repairs the other members, which ordered_json + // relocates when adding a key makes its vector grow + set_parent(m_data.m_value.object->operator[](it.key()) = it.value()); } } @@ -3999,9 +4021,6 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } // a nested object is merged: continue with its parent -#if JSON_DIAGNOSTICS - target->set_parents(); -#endif target = stack.back().target; first = stack.back().position; last = stack.back().last; @@ -4023,10 +4042,9 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec continue; } } - target->m_data.m_value.object->operator[](first.key()) = first.value(); -#if JSON_DIAGNOSTICS - target->m_data.m_value.object->operator[](first.key()).m_parent = target; -#endif + // set_parent() also repairs the other members, which ordered_json + // relocates when adding a key makes its vector grow + target->set_parent(target->m_data.m_value.object->operator[](first.key()) = first.value()); ++first; } } diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 3c5821d92..9091c1198 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -25701,6 +25701,27 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } + /// @brief restore the parent pointers after erasing from an object + /// ordered_json keeps its members in a vector, and erasing a member + /// re-constructs every member after it in place, which resets their + /// parent pointers + void set_parents_after_object_erase() + { +#if JSON_DIAGNOSTICS +#ifdef JSON_HEDLEY_MSVC_VERSION +#pragma warning(push ) +#pragma warning(disable : 4127) // ignore warning to replace if with if constexpr +#endif + if (detail::is_ordered_map::value) + { + set_parents(); + } +#ifdef JSON_HEDLEY_MSVC_VERSION +#pragma warning( pop ) +#endif +#endif + } + public: ////////////////////////// // JSON parser callback // @@ -27389,6 +27410,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec case value_t::object: { result.m_it.object_iterator = erase_from_object(pos.m_it.object_iterator); + set_parents_after_object_erase(); break; } @@ -27461,6 +27483,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec { result.m_it.object_iterator = m_data.m_value.object->erase(first.m_it.object_iterator, last.m_it.object_iterator); + set_parents_after_object_erase(); break; } @@ -27491,7 +27514,9 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_THROW(type_error::create(307, detail::concat("cannot use erase() with ", type_name()), this)); } - return m_data.m_value.object->erase(std::forward(key)); + const auto erased = m_data.m_value.object->erase(std::forward(key)); + set_parents_after_object_erase(); + return erased; } template < typename KeyType, detail::enable_if_t < @@ -27508,6 +27533,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec if (it != m_data.m_value.object->end()) { m_data.m_value.object->erase(it); + set_parents_after_object_erase(); return 1; } return 0; @@ -28419,16 +28445,12 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec if (it2 != m_data.m_value.object->end() && it2->second.is_object()) { it2->second.update_members(it.value().cbegin(), it.value().cend(), true, depth + 1); -#if JSON_DIAGNOSTICS - it2->second.set_parents(); -#endif continue; } } - m_data.m_value.object->operator[](it.key()) = it.value(); -#if JSON_DIAGNOSTICS - m_data.m_value.object->operator[](it.key()).m_parent = this; -#endif + // set_parent() also repairs the other members, which ordered_json + // relocates when adding a key makes its vector grow + set_parent(m_data.m_value.object->operator[](it.key()) = it.value()); } } @@ -28457,9 +28479,6 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } // a nested object is merged: continue with its parent -#if JSON_DIAGNOSTICS - target->set_parents(); -#endif target = stack.back().target; first = stack.back().position; last = stack.back().last; @@ -28481,10 +28500,9 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec continue; } } - target->m_data.m_value.object->operator[](first.key()) = first.value(); -#if JSON_DIAGNOSTICS - target->m_data.m_value.object->operator[](first.key()).m_parent = target; -#endif + // set_parent() also repairs the other members, which ordered_json + // relocates when adding a key makes its vector grow + target->set_parent(target->m_data.m_value.object->operator[](first.key()) = first.value()); ++first; } } diff --git a/tests/src/unit-diagnostics.cpp b/tests/src/unit-diagnostics.cpp index cdb6185b3..3ae649e5b 100644 --- a/tests/src/unit-diagnostics.cpp +++ b/tests/src/unit-diagnostics.cpp @@ -361,6 +361,106 @@ TEST_CASE("Regression tests for extended diagnostics") CHECK(p == o); } } + + SECTION("Regression test - erase() and update() must keep JSON_DIAGNOSTICS parent pointers of ordered_json members") + { + // ordered_json keeps its members in a vector: erasing a member + // re-constructs all members after it in place, and adding a key may + // reallocate the vector; both reset the parent pointers of the members + // that were moved + using nlohmann::ordered_json; + + const auto check_parents = [](const ordered_json & j) + { + // const access, so operator[] cannot repair the parent pointers + CHECK_THROWS_WITH_AS(j["z"]["x"].at(0), "[json.exception.type_error.304] (/z/x) cannot use at() with number", ordered_json::type_error); + + // must not trigger assert_invariant() in a debug/assert-enabled build + ordered_json const copy = j; // NOLINT(performance-unnecessary-copy-initialization) + CHECK(copy == j); + }; + + // erase(key) + { + ordered_json j = {{"a", 1}, {"z", {{"x", 1}}}}; + CHECK(j.erase("a") == 1); + check_parents(j); + } + + // erase(iterator) + { + ordered_json j = {{"a", 1}, {"z", {{"x", 1}}}}; + j.erase(j.begin()); + check_parents(j); + } + + // erase(iterator, iterator) + { + ordered_json j = {{"a", 1}, {"b", 2}, {"z", {{"x", 1}}}}; + j.erase(j.begin(), j.find("z")); + check_parents(j); + } + + // patch() removes via erase(iterator) + { + ordered_json j = {{"a", 1}, {"z", {{"x", 1}}}}; + j.patch_inplace(ordered_json::parse(R"([{"op": "remove", "path": "/a"}])")); + check_parents(j); + } + + // update(j) + { + ordered_json j = {{"z", {{"x", 1}}}}; + j.update({{"a", 1}, {"b", 2}}); + check_parents(j); + } + + // update(j, true), the outer and the nested vector both grow + { + ordered_json j = {{"z", {{"x", 1}}}}; + j.update({{"z", {{"y", 2}}}, {"a", 1}}, true); + check_parents(j); + } + + // update(j, true) around its descent bound, where the nested vectors + // grow while the objects are merged without recursing + for (const std::size_t depth : + { + nlohmann::detail::recursion_depth_limit() - 1, nlohmann::detail::recursion_depth_limit(), nlohmann::detail::recursion_depth_limit() + 2 + }) + { + ordered_json j = {{"z", {{"x", 1}}}}; + ordered_json patch = {{"a", 1}, {"b", 2}, {"c", {{"d", 3}}}}; + for (std::size_t i = 0; i < depth; ++i) + { + j = ordered_json{{"k", 0}, {"n", std::move(j)}}; + patch = ordered_json{{"n", std::move(patch)}, {"l", 1}, {"m", 2}}; + } + j.update(patch, true); + + // must not trigger assert_invariant() on any level in a + // debug/assert-enabled build + ordered_json const copy = j; // NOLINT(performance-unnecessary-copy-initialization) + CHECK(copy == j); + } + + // merge_patch() inserts "c" and removes "d" at /a/c, then inserts "e" + // at /a, which copies /a/c + { + auto j = ordered_json::parse(R"({"a": {"c": {"d": {}}}})"); + j.merge_patch(ordered_json::parse(R"({"a": {"c": {"c": "s", "d": null}, "e": "s"}})")); + CHECK(j.dump() == R"({"a":{"c":{"c":"s"},"e":"s"}})"); + + auto const& constJ = j; +#if JSON_DIAGNOSTIC_POSITIONS + CHECK_THROWS_WITH_AS(constJ["a"]["c"]["c"].at(0), "[json.exception.type_error.304] (/a/c/c) (bytes 18-21) cannot use at() with string", ordered_json::type_error); +#else + CHECK_THROWS_WITH_AS(constJ["a"]["c"]["c"].at(0), "[json.exception.type_error.304] (/a/c/c) cannot use at() with string", ordered_json::type_error); +#endif + ordered_json const copy = j; + CHECK(copy == j); + } + } } TEST_CASE("Better diagnostics past the descent bound of update() and merge_patch()") From 6c8ea0a6d1b36dfc9a11a83cb052c9436500d529 Mon Sep 17 00:00:00 2001 From: Jeremy Nimmer Date: Thu, 24 Sep 2026 23:46:19 -0700 Subject: [PATCH 40/64] Remove Bazel alwayslink=True (#5376) This should have no effect for header only libraries as mentioned. It was previously removed in e509007d but then accidentally added again in 26cfec34. Signed-off-by: Jeremy Nimmer --- BUILD.bazel | 1 - cmake/scripts/gen_bazel_build_file.cmake | 1 - 2 files changed, 2 deletions(-) diff --git a/BUILD.bazel b/BUILD.bazel index b13e62c22..db59095cd 100644 --- a/BUILD.bazel +++ b/BUILD.bazel @@ -71,7 +71,6 @@ cc_library( ], includes = ["include"], visibility = ["//visibility:public"], - alwayslink = True, ) cc_library( diff --git a/cmake/scripts/gen_bazel_build_file.cmake b/cmake/scripts/gen_bazel_build_file.cmake index 3c7db9493..a781a2e3f 100644 --- a/cmake/scripts/gen_bazel_build_file.cmake +++ b/cmake/scripts/gen_bazel_build_file.cmake @@ -42,7 +42,6 @@ string(APPEND CONTENT [=[ ], includes = ["include"], visibility = ["//visibility:public"], - alwayslink = True, ) cc_library( From 98278dc3f6899244cb5b3ffd07dd8c2379dcd4dd Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 25 Sep 2026 17:54:40 +0200 Subject: [PATCH 41/64] Fix CI: disable MSVC warning C5285 for the vendored doctest (#5577) The windows-11-arm runner now ships MSVC 19.51, which reports doctest's forward declaration of std::tuple as C5285 ("cannot declare a specialization for 'std::tuple'"). With /WX this breaks the msvc-arm64 job on develop and on every open pull request. Disable the warning for the test targets, like the other MSVC warnings already disabled there. Signed-off-by: Niels Lohmann --- tests/CMakeLists.txt | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/tests/CMakeLists.txt b/tests/CMakeLists.txt index 7a9bc5471..ea6da107c 100644 --- a/tests/CMakeLists.txt +++ b/tests/CMakeLists.txt @@ -80,7 +80,10 @@ target_compile_options(test_main PUBLIC # std::allocator_traits<...>::construct for the custom-base-class # test's map type past VS2015's limit. The name is only used for # debug info, so truncation does not affect the build. - $<$:/W4;/wd4566;/wd4996;/wd4702;/wd4503> + # Disable warning C5285: cannot declare a specialization for 'std::tuple'; MSVC 19.51 + # reports the forward declarations of standard library + # templates in the vendored doctest.h + $<$:/W4;/wd4566;/wd4996;/wd4702;/wd4503;/wd5285> # https://github.com/nlohmann/json/issues/1114 $<$:/bigobj> $<$:-Wa,-mbig-obj> From abbe52d6de81323fa0b159da2c36f0589756af16 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 25 Sep 2026 17:56:18 +0200 Subject: [PATCH 42/64] Add JSON_PRECISE_STREAM_POSITION to leave the character that terminates a number in the stream (#5344) * docs: qualify the operator>> stream positioning guarantee operator>>'s notes state that it leaves the stream positioned right after the parsed value, so that concatenated JSON values can be read back to back. That does not hold when the value is a number: a number is only terminated by the character that follows it, and the lexer's unget() is simulated (it rewinds only the lexer's own bookkeeping), so that character stays consumed from the stream. Document the actual behaviour: the guarantee holds for all value types except numbers, which must be followed by whitespace. Also qualify the cross-reference on the JSON Lines page, which repeated the unqualified claim. Documentation only; the behaviour itself is tracked in #5340. Signed-off-by: Niels Lohmann * fix: restore the character that terminates a number (#5340) operator>> is documented to leave the stream positioned right after the parsed value, so that concatenated JSON values can be read back to back. That did not hold for numbers: a number is only terminated by the character following it, and lexer::scan_number() reads that character and calls unget() -- which is simulated and rewinds only the lexer's own bookkeeping. input_stream_adapter consumes via sbumpc() with no matching sungetc(), so the terminating character stayed consumed and the next extraction started one byte too late ('1true' left the stream at 'rue'). Propagating unget() to the adapter directly does not work: next_unget makes the following get() replay the cached character, so the terminator would be delivered twice. Instead, restore the still-pending character once at the end of a non-strict parse, where the input is handed back to the caller: - input_stream_adapter gains unget_character() (sungetc()) and advertises it via supports_unget, detected the same way as supports_seek. - lexer::restore_pending_unget() turns a pending simulated unget of a real (non-EOF) character into a real one and clears next_unget so the character is not also replayed. It is a no-op for adapters that cannot unget, and reports failure when sungetc() fails, in which case the input is left as it was before. - parser calls it on the three non-strict paths, i.e. for operator>> and sax_parse(strict = false). Strict parse()/accept() are unaffected: they require the input to end after the value, so the character is consumed by the end-of-input check anyway. Parse error messages and reported positions are unchanged. Signed-off-by: Niels Lohmann * tests: fix CI failures in the #5340 test helpers Four CI failures, all in the new test code: - GCC (-Werror=useless-cast): drop the `json(...)` wrapper around `json::parse(...)`, which already returns a `json`. - GCC (-Werror=unused-result): assign the discarded `json::parse()` result to a dummy, the idiom used elsewhere in the test suite, and catch `json::parse_error&` for consistency. - clang-tidy (google-default-arguments): remove the default argument from the `pbackfail()` override; `sungetc()` supplies the base declaration's default. - MSVC (bad allocation): `no_putback_streambuf::underflow()` set a one-character get area without advancing `m_pos`, so an implementation whose `istream::get` peeks before it bumps re-read the same character forever. Keep no get area at all: `underflow()` peeks, `uflow()` consumes, and `sungetc()` still always lands in `pbackfail()`, which is what the test needs. Signed-off-by: Niels Lohmann * fix: leave the character that terminates a number in the input Read the character following a number without consuming it, instead of consuming it and putting it back. input_stream_adapter now peeks with sgetc() and only steps over the character when the next one is requested or when the adapter is destroyed, so releasing it cannot fail - no putback position is required from the streambuf. Suggested by gregmarr in #5344. Signed-off-by: Niels Lohmann * docs: match the version history wording to the peek-based fix Signed-off-by: Niels Lohmann * docs: drop the whitespace-separator caveat from the parsing pages The caveat added in #5343 describes the behavior this branch fixes: a number no longer consumes the character that terminates it, so concatenated values need no separator. Signed-off-by: Niels Lohmann * refactor: split the strict and non-strict paths in parser Folding the release_lookahead() call into the existing strict check left the "in strict mode" comment on an else-if branch, and made the strict condition in sax_parse() redundant with the branch it followed. Signed-off-by: Niels Lohmann * Put the stream position fix behind JSON_PRECISE_STREAM_POSITION Leaving the character that terminates a number in the stream is observable: reading "1,2,3" with repeated operator>> works today only because the comma after each number is swallowed, and std::getline after a number skips the line break. Both break with the fix, so make it opt-in for 3.x, as suggested by @gregmarr in the review. - JSON_PRECISE_STREAM_POSITION (default 0) selects the peek-based input_stream_adapter. Without it, the adapter is the consuming one from develop and has no supports_lookahead, so lexer::release_lookahead() and the parser's calls to it compile to nothing. - The macro changes input_stream_adapter's layout and member functions, so it gets the ABI tag _psp, after _bics. The ABI config tests, the natvis generator, and nlohmann_json.natvis (regenerated) know the tag. - The tests for the fix move to unit-precise-stream-position.cpp, which defines the macro itself and runs in every build, and gain the two cases above. unit-deserialization.cpp pins the default behavior instead. - The docs describe the default behavior again and point to the new macro page; version history says "added in 3.13.0, planned default in 4.0.0". Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- docs/mkdocs/docs/api/basic_json/sax_parse.md | 6 +- docs/mkdocs/docs/api/macros/index.md | 2 + .../macros/json_precise_stream_position.md | 131 +++ docs/mkdocs/docs/api/operator_gtgt.md | 7 +- docs/mkdocs/docs/features/macros.md | 9 + docs/mkdocs/docs/features/namespace.md | 1 + docs/mkdocs/docs/features/parsing/index.md | 3 +- docs/mkdocs/mkdocs.yml | 1 + include/nlohmann/detail/abi_macros.hpp | 19 +- .../nlohmann/detail/input/input_adapters.hpp | 82 ++ include/nlohmann/detail/input/lexer.hpp | 65 ++ include/nlohmann/detail/input/parser.hpp | 61 +- include/nlohmann/detail/macro_unscope.hpp | 1 + nlohmann_json.natvis | 960 ++++++++++++++++++ single_include/nlohmann/json.hpp | 228 ++++- single_include/nlohmann/json_fwd.hpp | 19 +- tests/abi/config/default.cpp | 4 + tests/abi/config/noversion.cpp | 4 + tests/src/unit-deserialization.cpp | 52 + tests/src/unit-precise-stream-position.cpp | 237 +++++ tools/generate_natvis/generate_natvis.py | 2 +- 21 files changed, 1846 insertions(+), 48 deletions(-) create mode 100644 docs/mkdocs/docs/api/macros/json_precise_stream_position.md create mode 100644 tests/src/unit-precise-stream-position.cpp diff --git a/docs/mkdocs/docs/api/basic_json/sax_parse.md b/docs/mkdocs/docs/api/basic_json/sax_parse.md index fc8ce07b6..bf61ea9eb 100644 --- a/docs/mkdocs/docs/api/basic_json/sax_parse.md +++ b/docs/mkdocs/docs/api/basic_json/sax_parse.md @@ -69,7 +69,9 @@ The SAX event lister must follow the interface of [`json_sax`](../json_sax/index [`input_format_t`](input_format_t.md) for more information `strict` (in) -: whether the input has to be consumed completely (optional, `#!cpp true` by default) +: whether the input has to be consumed completely (optional, `#!cpp true` by default); when `#!cpp false` and the + input is a `#!cpp std::istream`, the character that terminates a number is consumed unless + [`JSON_PRECISE_STREAM_POSITION`](../macros/json_precise_stream_position.md) is defined to `1`; see [`operator>>`](../operator_gtgt.md#notes) `ignore_comments` (in) : whether comments should be ignored and treated like whitespace (`#!cpp true`) or yield a parse error @@ -136,6 +138,8 @@ A UTF-8 byte order mark is silently ignored. - Added `ignore_trailing_commas` in version 3.13.0. - Extended container support (1) to include types with lvalue-only ADL `begin`/`end` (matching `std::begin`/`std::end` semantics) in version 3.13.0. - Extended overload (2) to accept heterogeneous iterator+sentinel pairs (C++20 ranges support) in version 3.13.0. +- `JSON_PRECISE_STREAM_POSITION` added in version 3.13.0 to optionally leave a `#!cpp std::istream` positioned right + after the parsed value when `strict` is `#!cpp false`. !!! warning "Deprecation" diff --git a/docs/mkdocs/docs/api/macros/index.md b/docs/mkdocs/docs/api/macros/index.md index 70f02a7a6..bf773b5c4 100644 --- a/docs/mkdocs/docs/api/macros/index.md +++ b/docs/mkdocs/docs/api/macros/index.md @@ -16,6 +16,8 @@ header. See also the [macro overview page](../../features/macros.md). ## Parsing +- [**JSON_PRECISE_STREAM_POSITION**](json_precise_stream_position.md) - opt in to leaving an input stream positioned + right after a parsed number - [**JSON_STRICT_NUL_HANDLING**](json_strict_nul_handling.md) - opt in to rejecting a NUL byte in the input instead of treating it as end of input diff --git a/docs/mkdocs/docs/api/macros/json_precise_stream_position.md b/docs/mkdocs/docs/api/macros/json_precise_stream_position.md new file mode 100644 index 000000000..5710e9975 --- /dev/null +++ b/docs/mkdocs/docs/api/macros/json_precise_stream_position.md @@ -0,0 +1,131 @@ +# JSON_PRECISE_STREAM_POSITION + +```cpp +#define JSON_PRECISE_STREAM_POSITION /* value */ +``` + +When defined to `1`, [`operator>>`](../operator_gtgt.md) and [`sax_parse`](../basic_json/sax_parse.md) with +`strict = false` leave a `#!cpp std::istream` positioned right after the parsed value for every value type. By default, +the character that terminates a number is consumed as well. + +The macro only affects reading from a `#!cpp std::istream` when the rest of the stream is not required to be consumed. +[`parse`](../basic_json/parse.md), [`accept`](../basic_json/accept.md), and all other inputs (strings, iterators, +containers, `#!cpp FILE*`) are never affected. + +## Default definition + +The default value is `0` (disabled — existing behavior is preserved). + +```cpp +#define JSON_PRECISE_STREAM_POSITION 0 +``` + +## Notes + +!!! note "Background" + + A number is the only JSON value whose end can be detected solely by reading the character that follows it. By + default, that character is consumed and not put back, so the stream is left one byte too far after a number, and + only after a number: + + ```cpp + std::istringstream input("1true"); + json j; + input >> j; // j == 1, but the stream now starts at "rue" + ``` + + With this macro, the character is only looked at and left in the stream, so the stream starts at `true`. This + does not require the stream buffer to support putting a character back. + + This was not changed unconditionally, because code can depend on the consumed character, even unknowingly (see + [#5340](https://github.com/nlohmann/json/issues/5340)). Both of the following work by default only because the + character after each number is swallowed, and behave differently with this macro: + + ```cpp + std::istringstream input("1,2,3"); + json j1, j2, j3; + input >> j1 >> j2 >> j3; // default: 1, 2, 3 + // with the macro: throws parse_error.101 at the ',' + ``` + + ```cpp + std::istringstream input("42\nfoo"); + json j; + std::string line; + input >> j; + std::getline(input, line); // default: "foo" + // with the macro: "" (like after reading an int with >>) + ``` + + In both cases, the behavior with the macro is what you already get today when the value is not a number: `"a","b"` + fails at the `,`, and `std::getline` after `{}` returns an empty string. This macro offers an opt-in path to + the consistent behavior ahead of version 4.0.0, where it is planned to become the default. + +!!! warning "Opt-in only" + + This macro must be defined **before** including ``. Defining it after the include has no + effect. + +!!! note "ABI compatibility" + + The value of this macro is encoded in the [namespace](../../features/namespace.md) (tag `_psp`), resulting in + distinct symbol names. Translation units compiled with and without it can therefore be linked into the same program + without One Definition Rule (ODR) violations, but they cannot exchange instances of library types. + +!!! tip "Workaround without the macro" + + Separate the values in the stream with whitespace. The character consumed after a number is then the separator, + and whitespace before the next value is skipped anyway. + +## Examples + +??? example "Default behavior (macro not defined)" + + Without the macro, the character after a number is consumed: + + ```cpp + #include + #include + #include + + using json = nlohmann::json; + + int main() + { + std::istringstream input("1true"); + json j1, j2; + input >> j1; // j1 == 1 + input >> j2; // throws parse_error.101: the stream now starts at "rue" + } + ``` + +??? example "Opt-in precise stream position (macro defined to 1)" + + With the macro, the stream is positioned right after the number: + + ```cpp + #define JSON_PRECISE_STREAM_POSITION 1 + #include + #include + #include + + using json = nlohmann::json; + + int main() + { + std::istringstream input("1true"); + json j1, j2; + input >> j1; // j1 == 1 + input >> j2; // j2 == true + } + ``` + +## See also + +- [**operator>>**](../operator_gtgt.md) - deserialize from stream +- [**sax_parse**](../basic_json/sax_parse.md) - generate SAX events + +## Version history + +- Added in version 3.13.0. +- Planned to become the default (with the macro removed) in version 4.0.0. diff --git a/docs/mkdocs/docs/api/operator_gtgt.md b/docs/mkdocs/docs/api/operator_gtgt.md index 3e60d5236..0173b9fb3 100644 --- a/docs/mkdocs/docs/api/operator_gtgt.md +++ b/docs/mkdocs/docs/api/operator_gtgt.md @@ -67,7 +67,9 @@ input >> j2; // parses the next value Only numbers are affected. Values ending in a self-delimiting character do not read past themselves, so `truefalse`, `[1][2]`, `{"a":1}{"b":2}`, and `"a""b"` can be read back to back without a separator. - This is tracked in [#5340](https://github.com/nlohmann/json/issues/5340). + Define [`JSON_PRECISE_STREAM_POSITION`](macros/json_precise_stream_position.md) to `1` to leave the terminating character in the stream + instead, so that the stream is positioned right after the value for every value type and no separator is + needed. This is tracked in [#5340](https://github.com/nlohmann/json/issues/5340). Note that reading concatenated values does **not** work for [JSON Lines](../features/parsing/json_lines.md) (newline-delimited JSON) input -- see that page for why and for the recommended alternative. @@ -107,9 +109,12 @@ being read. - [parse](basic_json/parse.md) - deserialize from a compatible input - [`JSON_STRICT_NUL_HANDLING`](macros/json_strict_nul_handling.md) - opt in to rejecting a NUL byte in the input instead of treating it as end of input +- [`JSON_PRECISE_STREAM_POSITION`](macros/json_precise_stream_position.md) - opt in to leaving the stream positioned right after a number ## Version history - Added in version 1.0.0. - `JSON_STRICT_NUL_HANDLING` added in version 3.13.0 to optionally reject a NUL byte in the input instead of treating it as end of input; planned to become the default in version 4.0.0. +- `JSON_PRECISE_STREAM_POSITION` added in version 3.13.0 to optionally leave the character that terminates a number in + the stream; planned to become the default in version 4.0.0. diff --git a/docs/mkdocs/docs/features/macros.md b/docs/mkdocs/docs/features/macros.md index e7baba0ae..7d6d23148 100644 --- a/docs/mkdocs/docs/features/macros.md +++ b/docs/mkdocs/docs/features/macros.md @@ -98,6 +98,15 @@ rather than descending into a bounded number of levels first, which is slower bu See [full documentation of `JSON_NO_THREAD_LOCAL`](../api/macros/json_no_thread_local.md). +## `JSON_PRECISE_STREAM_POSITION` + +When defined to `1`, [`operator>>`](../api/operator_gtgt.md) and non-strict +[`sax_parse`](../api/basic_json/sax_parse.md) leave an input stream positioned right after the parsed value, instead of +also consuming the character that terminates a number. The default value is `0`, which preserves the existing behavior; +this is planned to become the default in version 4.0.0. + +See [full documentation of `JSON_PRECISE_STREAM_POSITION`](../api/macros/json_precise_stream_position.md). + ## `JSON_SKIP_LIBRARY_VERSION_CHECK` When defined, the library will not create a compiler warning when a different version of the library was already diff --git a/docs/mkdocs/docs/features/namespace.md b/docs/mkdocs/docs/features/namespace.md index 09e53f3a2..5eb4a76a9 100644 --- a/docs/mkdocs/docs/features/namespace.md +++ b/docs/mkdocs/docs/features/namespace.md @@ -18,6 +18,7 @@ The complete default namespace name is derived as follows: - [`JSON_DIAGNOSTIC_POSITIONS`](../api/macros/json_diagnostic_positions.md) defined non-zero appends `_dp`. - [`JSON_BRACE_INIT_COPY_SEMANTICS`](../api/macros/json_brace_init_copy_semantics.md) defined non-zero appends `_bics`. + - [`JSON_PRECISE_STREAM_POSITION`](../api/macros/json_precise_stream_position.md) defined non-zero appends `_psp`. - The inline namespace ends with the suffix `_v` followed by the 3 components of the version number separated by underscores. To omit the version component, see [Disabling the version component](#disabling-the-version-component) below. diff --git a/docs/mkdocs/docs/features/parsing/index.md b/docs/mkdocs/docs/features/parsing/index.md index 17624b8a2..476f024fa 100644 --- a/docs/mkdocs/docs/features/parsing/index.md +++ b/docs/mkdocs/docs/features/parsing/index.md @@ -41,7 +41,8 @@ document followed by trailing bytes" is accepted rather than rejected. If you ar reject any input that is not exactly one JSON document, prefer `parse`. When using `operator>>` to read several concatenated values this way, a value that is a number must be followed by -whitespace, because `operator>>` consumes the character that terminates a number — see the +whitespace, because `operator>>` consumes the character that terminates a number, unless +[`JSON_PRECISE_STREAM_POSITION`](../../api/macros/json_precise_stream_position.md) is defined to `1` — see the [`operator>>` notes](../../api/operator_gtgt.md#notes) for details and examples. ## SAX vs. DOM parsing diff --git a/docs/mkdocs/mkdocs.yml b/docs/mkdocs/mkdocs.yml index ffb0fae80..a05ce2dff 100644 --- a/docs/mkdocs/mkdocs.yml +++ b/docs/mkdocs/mkdocs.yml @@ -293,6 +293,7 @@ nav: - 'JSON_NOEXCEPTION': api/macros/json_noexception.md - 'JSON_NO_IO': api/macros/json_no_io.md - 'JSON_NO_THREAD_LOCAL': api/macros/json_no_thread_local.md + - 'JSON_PRECISE_STREAM_POSITION': api/macros/json_precise_stream_position.md - 'JSON_SKIP_LIBRARY_VERSION_CHECK': api/macros/json_skip_library_version_check.md - 'JSON_SKIP_UNSUPPORTED_COMPILER_CHECK': api/macros/json_skip_unsupported_compiler_check.md - 'JSON_STRICT_NUL_HANDLING': api/macros/json_strict_nul_handling.md diff --git a/include/nlohmann/detail/abi_macros.hpp b/include/nlohmann/detail/abi_macros.hpp index cca04e8ec..a6666c66e 100644 --- a/include/nlohmann/detail/abi_macros.hpp +++ b/include/nlohmann/detail/abi_macros.hpp @@ -38,6 +38,10 @@ #define JSON_BRACE_INIT_COPY_SEMANTICS 0 #endif +#ifndef JSON_PRECISE_STREAM_POSITION + #define JSON_PRECISE_STREAM_POSITION 0 +#endif + #if JSON_DIAGNOSTICS #define NLOHMANN_JSON_ABI_TAG_DIAGNOSTICS _diag #else @@ -62,21 +66,28 @@ #define NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS #endif +#if JSON_PRECISE_STREAM_POSITION + #define NLOHMANN_JSON_ABI_TAG_PRECISE_STREAM_POSITION _psp +#else + #define NLOHMANN_JSON_ABI_TAG_PRECISE_STREAM_POSITION +#endif + #ifndef NLOHMANN_JSON_NAMESPACE_NO_VERSION #define NLOHMANN_JSON_NAMESPACE_NO_VERSION 0 #endif // Construct the namespace ABI tags component -#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d) json_abi ## a ## b ## c ## d -#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d) \ - NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d) +#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e) json_abi ## a ## b ## c ## d ## e +#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d, e) \ + NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e) #define NLOHMANN_JSON_ABI_TAGS \ NLOHMANN_JSON_ABI_TAGS_CONCAT( \ NLOHMANN_JSON_ABI_TAG_DIAGNOSTICS, \ NLOHMANN_JSON_ABI_TAG_LEGACY_DISCARDED_VALUE_COMPARISON, \ NLOHMANN_JSON_ABI_TAG_DIAGNOSTIC_POSITIONS, \ - NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS) + NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS, \ + NLOHMANN_JSON_ABI_TAG_PRECISE_STREAM_POSITION) // Construct the namespace version component #define NLOHMANN_JSON_NAMESPACE_VERSION_CONCAT_EX(major, minor, patch) \ diff --git a/include/nlohmann/detail/input/input_adapters.hpp b/include/nlohmann/detail/input/input_adapters.hpp index 775a8398c..317b38806 100644 --- a/include/nlohmann/detail/input/input_adapters.hpp +++ b/include/nlohmann/detail/input/input_adapters.hpp @@ -101,6 +101,11 @@ class input_stream_adapter // maintain ifstream flags, except eof if (is != nullptr) { +#if JSON_PRECISE_STREAM_POSITION + // consume the character last returned by get_character() unless it + // was given back with release_lookahead() + commit_lookahead(); +#endif is->clear(is->rdstate() & std::ios::eofbit); } } @@ -114,6 +119,58 @@ class input_stream_adapter input_stream_adapter& operator=(input_stream_adapter&) = delete; input_stream_adapter& operator=(input_stream_adapter&&) = delete; +#if JSON_PRECISE_STREAM_POSITION + input_stream_adapter(input_stream_adapter&& rhs) noexcept + : is(rhs.is), sb(rhs.sb), lookahead(rhs.lookahead) + { + rhs.is = nullptr; + rhs.sb = nullptr; + rhs.lookahead = false; + } + + // Whether the character last returned by get_character() can be given back + // to the input with release_lookahead(). + static constexpr bool supports_lookahead = true; + + // std::istream/std::streambuf use std::char_traits::to_int_type, to + // ensure that std::char_traits::eof() and the character 0xFF do not + // end up as the same value, e.g., 0xFFFFFFFF. + // + // The character is peeked rather than consumed: it is only stepped over + // once the next character is requested, or when the adapter is destroyed. + // Until then, release_lookahead() can leave it in the input. + std::char_traits::int_type get_character() + { + if (lookahead) + { + // step over the character returned by the previous call + sb->sbumpc(); + } + + auto res = sb->sgetc(); + // set eof manually, as we don't use the istream interface. + if (JSON_HEDLEY_UNLIKELY(res == std::char_traits::eof())) + { + // there is nothing to step over next time + lookahead = false; + is->clear(is->rdstate() | std::ios::eofbit); + } + else + { + lookahead = true; + } + return res; + } + + // Leave the character last returned by get_character() in the input, so + // that the next read from the stream - by this adapter or by the caller + // once parsing is done - sees it again. Unlike putting a consumed + // character back, this cannot fail. + void release_lookahead() noexcept + { + lookahead = false; + } +#else input_stream_adapter(input_stream_adapter&& rhs) noexcept : is(rhs.is), sb(rhs.sb) { @@ -124,6 +181,9 @@ class input_stream_adapter // std::istream/std::streambuf use std::char_traits::to_int_type, to // ensure that std::char_traits::eof() and the character 0xFF do not // end up as the same value, e.g., 0xFFFFFFFF. + // + // The character is consumed, so the character that terminates a number + // stays consumed after parsing; see JSON_PRECISE_STREAM_POSITION. std::char_traits::int_type get_character() { auto res = sb->sbumpc(); @@ -134,10 +194,14 @@ class input_stream_adapter } return res; } +#endif template std::size_t get_elements(T* dest, std::size_t count = 1) { +#if JSON_PRECISE_STREAM_POSITION + commit_lookahead(); +#endif auto res = static_cast(sb->sgetn(reinterpret_cast(dest), static_cast(count * sizeof(T)))); if (JSON_HEDLEY_UNLIKELY(res < count * sizeof(T))) { @@ -147,9 +211,27 @@ class input_stream_adapter } private: +#if JSON_PRECISE_STREAM_POSITION + // Step over the character last returned by get_character(). The character + // has already been peeked successfully, so for every streambuf with a get + // area this is a pointer increment that cannot fail. + void commit_lookahead() + { + if (lookahead) + { + lookahead = false; + sb->sbumpc(); + } + } +#endif + /// the associated input stream std::istream* is = nullptr; std::streambuf* sb = nullptr; +#if JSON_PRECISE_STREAM_POSITION + /// whether get_character() peeked a character that is not consumed yet + bool lookahead = false; +#endif }; #endif // JSON_NO_IO diff --git a/include/nlohmann/detail/input/lexer.hpp b/include/nlohmann/detail/input/lexer.hpp index fe85cd53d..98c0fd76a 100644 --- a/include/nlohmann/detail/input/lexer.hpp +++ b/include/nlohmann/detail/input/lexer.hpp @@ -127,6 +127,25 @@ constexpr bool input_adapter_supports_seek(std::false_type /*detected*/) return false; } +// Detect whether an input adapter reads with one character of lookahead that +// can be left in the input (see input_stream_adapter::supports_lookahead, +// which is only defined with JSON_PRECISE_STREAM_POSITION), detected like +// supports_seek above. +template +using detect_supports_lookahead = decltype(InputAdapterType::supports_lookahead); + +template +constexpr bool input_adapter_supports_lookahead(std::true_type /*detected*/) +{ + return InputAdapterType::supports_lookahead; +} + +template +constexpr bool input_adapter_supports_lookahead(std::false_type /*detected*/) +{ + return false; +} + // Detect whether an input adapter exposes a contiguous byte block that the // lexer can scan directly (see iterator_input_adapter::supports_bulk_scan). // Adapters without the flag - file, stream, wide-string, user-defined - fall @@ -167,6 +186,12 @@ class lexer : public lexer_base static constexpr bool lazy_token_string = input_adapter_supports_seek(is_detected {}); + /// whether a simulated unget can be passed on to the input adapter, which + /// then leaves the character in the input; see + /// input_adapter_supports_lookahead + static constexpr bool can_release_lookahead = + input_adapter_supports_lookahead(is_detected {}); + /// whether string scanning may bulk-consume runs of ordinary characters /// directly from a contiguous input buffer (SWAR fast path). This requires /// the token to be reconstructible lazily (lazy_token_string), so bypassing @@ -1898,6 +1923,21 @@ scan_number_done: uncapture_char(std::integral_constant {}); } + /// adapter without lookahead: nothing to do (see release_lookahead) + void release_lookahead_impl(std::false_type /*can_release*/) const noexcept {} + + /// adapter with lookahead: leave the character in the input instead + void release_lookahead_impl(std::true_type /*can_release*/) + { + if (next_unget) + { + // the character is read from the input again rather than replayed + // from current, so the adapter must not step over it + next_unget = false; + ia.release_lookahead(); + } + } + /// seekable adapter: nothing was captured, so nothing to undo void uncapture_char(std::true_type /*lazy*/) const noexcept {} @@ -1961,6 +2001,31 @@ scan_number_done: return position; } + /*! + @brief pass a pending simulated unget on to the input + + unget() only rewinds the lexer's own bookkeeping, so the character that + terminated the last token (e.g. the character after a number) would still + be stepped over when the input adapter is done. Callers that hand the + input back to the user afterwards - operator>> and non-strict sax_parse - + call this once when scanning is done, so that the input is positioned + right after the value. + + Adapters without lookahead (see input_adapter_supports_lookahead) are not + handed back to the user, so this is a no-op for them. Without + JSON_PRECISE_STREAM_POSITION, no adapter has lookahead, so this is always a + no-op and the terminating character stays consumed. + + Scanning may continue after this call: @a next_unget is cleared, and the + character is read from the input again instead of being replayed from + @a current. A pending unget of EOF needs no special case, because reaching + EOF leaves no lookahead to release. + */ + void release_lookahead() + { + release_lookahead_impl(std::integral_constant {}); + } + #if JSON_DIAGNOSTIC_POSITIONS /// return the offset of the first character of the last read token; unlike /// the token's parsed value, this accounts for escape sequences diff --git a/include/nlohmann/detail/input/parser.hpp b/include/nlohmann/detail/input/parser.hpp index a45ee4a0a..5fec57a70 100644 --- a/include/nlohmann/detail/input/parser.hpp +++ b/include/nlohmann/detail/input/parser.hpp @@ -100,13 +100,22 @@ class parser json_sax_dom_callback_parser sdp(result, callback, allow_exceptions, &m_lexer); sax_parse_internal(&sdp); - // in strict mode, input must be completely read - if (strict && (get_token() != token_type::end_of_input)) + if (strict) { - sdp.parse_error(m_lexer.get_position(), - m_lexer.get_token_string(), - parse_error::create(101, m_lexer.get_position(), - exception_message(token_type::end_of_input, "value"), nullptr)); + // in strict mode, input must be completely read + if (get_token() != token_type::end_of_input) + { + sdp.parse_error(m_lexer.get_position(), + m_lexer.get_token_string(), + parse_error::create(101, m_lexer.get_position(), + exception_message(token_type::end_of_input, "value"), nullptr)); + } + } + else + { + // the caller keeps using the input: position it right after + // the value by leaving the character that terminated it + m_lexer.release_lookahead(); } // in case of an error, return a discarded value @@ -128,12 +137,20 @@ class parser json_sax_dom_parser sdp(result, allow_exceptions, &m_lexer); sax_parse_internal(&sdp); - // in strict mode, input must be completely read - if (strict && (get_token() != token_type::end_of_input)) + if (strict) { - sdp.parse_error(m_lexer.get_position(), - m_lexer.get_token_string(), - parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_of_input, "value"), nullptr)); + // in strict mode, input must be completely read + if (get_token() != token_type::end_of_input) + { + sdp.parse_error(m_lexer.get_position(), + m_lexer.get_token_string(), + parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_of_input, "value"), nullptr)); + } + } + else + { + // see above + m_lexer.release_lookahead(); } // in case of an error, return a discarded value @@ -166,12 +183,24 @@ class parser (void)detail::is_sax_static_asserts {}; const bool result = sax_parse_internal(sax); - // strict mode: next byte must be EOF - if (result && strict && (get_token() != token_type::end_of_input)) + if (result) { - return sax->parse_error(m_lexer.get_position(), - m_lexer.get_token_string(), - parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_of_input, "value"), nullptr)); + if (strict) + { + // strict mode: next byte must be EOF + if (get_token() != token_type::end_of_input) + { + return sax->parse_error(m_lexer.get_position(), + m_lexer.get_token_string(), + parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_of_input, "value"), nullptr)); + } + } + else + { + // the caller keeps using the input: position it right after + // the value by leaving the character that terminated it + m_lexer.release_lookahead(); + } } return result; diff --git a/include/nlohmann/detail/macro_unscope.hpp b/include/nlohmann/detail/macro_unscope.hpp index afcbfc38b..6ace4cf4a 100644 --- a/include/nlohmann/detail/macro_unscope.hpp +++ b/include/nlohmann/detail/macro_unscope.hpp @@ -45,6 +45,7 @@ #undef JSON_HAS_STATIC_RTTI #undef JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON #undef JSON_BRACE_INIT_COPY_SEMANTICS + #undef JSON_PRECISE_STREAM_POSITION #endif #include diff --git a/nlohmann_json.natvis b/nlohmann_json.natvis index 2eccbe17c..8f4eec31a 100644 --- a/nlohmann_json.natvis +++ b/nlohmann_json.natvis @@ -335,6 +335,66 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -515,6 +575,66 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -635,6 +755,66 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -695,6 +875,126 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -815,6 +1115,66 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -875,6 +1235,126 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -935,6 +1415,186 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -995,4 +1655,304 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 9091c1198..e3ca0f0d0 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -95,6 +95,10 @@ #define JSON_BRACE_INIT_COPY_SEMANTICS 0 #endif +#ifndef JSON_PRECISE_STREAM_POSITION + #define JSON_PRECISE_STREAM_POSITION 0 +#endif + #if JSON_DIAGNOSTICS #define NLOHMANN_JSON_ABI_TAG_DIAGNOSTICS _diag #else @@ -119,21 +123,28 @@ #define NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS #endif +#if JSON_PRECISE_STREAM_POSITION + #define NLOHMANN_JSON_ABI_TAG_PRECISE_STREAM_POSITION _psp +#else + #define NLOHMANN_JSON_ABI_TAG_PRECISE_STREAM_POSITION +#endif + #ifndef NLOHMANN_JSON_NAMESPACE_NO_VERSION #define NLOHMANN_JSON_NAMESPACE_NO_VERSION 0 #endif // Construct the namespace ABI tags component -#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d) json_abi ## a ## b ## c ## d -#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d) \ - NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d) +#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e) json_abi ## a ## b ## c ## d ## e +#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d, e) \ + NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e) #define NLOHMANN_JSON_ABI_TAGS \ NLOHMANN_JSON_ABI_TAGS_CONCAT( \ NLOHMANN_JSON_ABI_TAG_DIAGNOSTICS, \ NLOHMANN_JSON_ABI_TAG_LEGACY_DISCARDED_VALUE_COMPARISON, \ NLOHMANN_JSON_ABI_TAG_DIAGNOSTIC_POSITIONS, \ - NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS) + NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS, \ + NLOHMANN_JSON_ABI_TAG_PRECISE_STREAM_POSITION) // Construct the namespace version component #define NLOHMANN_JSON_NAMESPACE_VERSION_CONCAT_EX(major, minor, patch) \ @@ -7419,6 +7430,11 @@ class input_stream_adapter // maintain ifstream flags, except eof if (is != nullptr) { +#if JSON_PRECISE_STREAM_POSITION + // consume the character last returned by get_character() unless it + // was given back with release_lookahead() + commit_lookahead(); +#endif is->clear(is->rdstate() & std::ios::eofbit); } } @@ -7432,6 +7448,58 @@ class input_stream_adapter input_stream_adapter& operator=(input_stream_adapter&) = delete; input_stream_adapter& operator=(input_stream_adapter&&) = delete; +#if JSON_PRECISE_STREAM_POSITION + input_stream_adapter(input_stream_adapter&& rhs) noexcept + : is(rhs.is), sb(rhs.sb), lookahead(rhs.lookahead) + { + rhs.is = nullptr; + rhs.sb = nullptr; + rhs.lookahead = false; + } + + // Whether the character last returned by get_character() can be given back + // to the input with release_lookahead(). + static constexpr bool supports_lookahead = true; + + // std::istream/std::streambuf use std::char_traits::to_int_type, to + // ensure that std::char_traits::eof() and the character 0xFF do not + // end up as the same value, e.g., 0xFFFFFFFF. + // + // The character is peeked rather than consumed: it is only stepped over + // once the next character is requested, or when the adapter is destroyed. + // Until then, release_lookahead() can leave it in the input. + std::char_traits::int_type get_character() + { + if (lookahead) + { + // step over the character returned by the previous call + sb->sbumpc(); + } + + auto res = sb->sgetc(); + // set eof manually, as we don't use the istream interface. + if (JSON_HEDLEY_UNLIKELY(res == std::char_traits::eof())) + { + // there is nothing to step over next time + lookahead = false; + is->clear(is->rdstate() | std::ios::eofbit); + } + else + { + lookahead = true; + } + return res; + } + + // Leave the character last returned by get_character() in the input, so + // that the next read from the stream - by this adapter or by the caller + // once parsing is done - sees it again. Unlike putting a consumed + // character back, this cannot fail. + void release_lookahead() noexcept + { + lookahead = false; + } +#else input_stream_adapter(input_stream_adapter&& rhs) noexcept : is(rhs.is), sb(rhs.sb) { @@ -7442,6 +7510,9 @@ class input_stream_adapter // std::istream/std::streambuf use std::char_traits::to_int_type, to // ensure that std::char_traits::eof() and the character 0xFF do not // end up as the same value, e.g., 0xFFFFFFFF. + // + // The character is consumed, so the character that terminates a number + // stays consumed after parsing; see JSON_PRECISE_STREAM_POSITION. std::char_traits::int_type get_character() { auto res = sb->sbumpc(); @@ -7452,10 +7523,14 @@ class input_stream_adapter } return res; } +#endif template std::size_t get_elements(T* dest, std::size_t count = 1) { +#if JSON_PRECISE_STREAM_POSITION + commit_lookahead(); +#endif auto res = static_cast(sb->sgetn(reinterpret_cast(dest), static_cast(count * sizeof(T)))); if (JSON_HEDLEY_UNLIKELY(res < count * sizeof(T))) { @@ -7465,9 +7540,27 @@ class input_stream_adapter } private: +#if JSON_PRECISE_STREAM_POSITION + // Step over the character last returned by get_character(). The character + // has already been peeked successfully, so for every streambuf with a get + // area this is a pointer increment that cannot fail. + void commit_lookahead() + { + if (lookahead) + { + lookahead = false; + sb->sbumpc(); + } + } +#endif + /// the associated input stream std::istream* is = nullptr; std::streambuf* sb = nullptr; +#if JSON_PRECISE_STREAM_POSITION + /// whether get_character() peeked a character that is not consumed yet + bool lookahead = false; +#endif }; #endif // JSON_NO_IO @@ -8879,6 +8972,25 @@ constexpr bool input_adapter_supports_seek(std::false_type /*detected*/) return false; } +// Detect whether an input adapter reads with one character of lookahead that +// can be left in the input (see input_stream_adapter::supports_lookahead, +// which is only defined with JSON_PRECISE_STREAM_POSITION), detected like +// supports_seek above. +template +using detect_supports_lookahead = decltype(InputAdapterType::supports_lookahead); + +template +constexpr bool input_adapter_supports_lookahead(std::true_type /*detected*/) +{ + return InputAdapterType::supports_lookahead; +} + +template +constexpr bool input_adapter_supports_lookahead(std::false_type /*detected*/) +{ + return false; +} + // Detect whether an input adapter exposes a contiguous byte block that the // lexer can scan directly (see iterator_input_adapter::supports_bulk_scan). // Adapters without the flag - file, stream, wide-string, user-defined - fall @@ -8919,6 +9031,12 @@ class lexer : public lexer_base static constexpr bool lazy_token_string = input_adapter_supports_seek(is_detected {}); + /// whether a simulated unget can be passed on to the input adapter, which + /// then leaves the character in the input; see + /// input_adapter_supports_lookahead + static constexpr bool can_release_lookahead = + input_adapter_supports_lookahead(is_detected {}); + /// whether string scanning may bulk-consume runs of ordinary characters /// directly from a contiguous input buffer (SWAR fast path). This requires /// the token to be reconstructible lazily (lazy_token_string), so bypassing @@ -10650,6 +10768,21 @@ scan_number_done: uncapture_char(std::integral_constant {}); } + /// adapter without lookahead: nothing to do (see release_lookahead) + void release_lookahead_impl(std::false_type /*can_release*/) const noexcept {} + + /// adapter with lookahead: leave the character in the input instead + void release_lookahead_impl(std::true_type /*can_release*/) + { + if (next_unget) + { + // the character is read from the input again rather than replayed + // from current, so the adapter must not step over it + next_unget = false; + ia.release_lookahead(); + } + } + /// seekable adapter: nothing was captured, so nothing to undo void uncapture_char(std::true_type /*lazy*/) const noexcept {} @@ -10713,6 +10846,31 @@ scan_number_done: return position; } + /*! + @brief pass a pending simulated unget on to the input + + unget() only rewinds the lexer's own bookkeeping, so the character that + terminated the last token (e.g. the character after a number) would still + be stepped over when the input adapter is done. Callers that hand the + input back to the user afterwards - operator>> and non-strict sax_parse - + call this once when scanning is done, so that the input is positioned + right after the value. + + Adapters without lookahead (see input_adapter_supports_lookahead) are not + handed back to the user, so this is a no-op for them. Without + JSON_PRECISE_STREAM_POSITION, no adapter has lookahead, so this is always a + no-op and the terminating character stays consumed. + + Scanning may continue after this call: @a next_unget is cleared, and the + character is read from the input again instead of being replayed from + @a current. A pending unget of EOF needs no special case, because reaching + EOF leaves no lookahead to release. + */ + void release_lookahead() + { + release_lookahead_impl(std::integral_constant {}); + } + #if JSON_DIAGNOSTIC_POSITIONS /// return the offset of the first character of the last read token; unlike /// the token's parsed value, this accounts for escape sequences @@ -15960,13 +16118,22 @@ class parser json_sax_dom_callback_parser sdp(result, callback, allow_exceptions, &m_lexer); sax_parse_internal(&sdp); - // in strict mode, input must be completely read - if (strict && (get_token() != token_type::end_of_input)) + if (strict) { - sdp.parse_error(m_lexer.get_position(), - m_lexer.get_token_string(), - parse_error::create(101, m_lexer.get_position(), - exception_message(token_type::end_of_input, "value"), nullptr)); + // in strict mode, input must be completely read + if (get_token() != token_type::end_of_input) + { + sdp.parse_error(m_lexer.get_position(), + m_lexer.get_token_string(), + parse_error::create(101, m_lexer.get_position(), + exception_message(token_type::end_of_input, "value"), nullptr)); + } + } + else + { + // the caller keeps using the input: position it right after + // the value by leaving the character that terminated it + m_lexer.release_lookahead(); } // in case of an error, return a discarded value @@ -15988,12 +16155,20 @@ class parser json_sax_dom_parser sdp(result, allow_exceptions, &m_lexer); sax_parse_internal(&sdp); - // in strict mode, input must be completely read - if (strict && (get_token() != token_type::end_of_input)) + if (strict) { - sdp.parse_error(m_lexer.get_position(), - m_lexer.get_token_string(), - parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_of_input, "value"), nullptr)); + // in strict mode, input must be completely read + if (get_token() != token_type::end_of_input) + { + sdp.parse_error(m_lexer.get_position(), + m_lexer.get_token_string(), + parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_of_input, "value"), nullptr)); + } + } + else + { + // see above + m_lexer.release_lookahead(); } // in case of an error, return a discarded value @@ -16026,12 +16201,24 @@ class parser (void)detail::is_sax_static_asserts {}; const bool result = sax_parse_internal(sax); - // strict mode: next byte must be EOF - if (result && strict && (get_token() != token_type::end_of_input)) + if (result) { - return sax->parse_error(m_lexer.get_position(), - m_lexer.get_token_string(), - parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_of_input, "value"), nullptr)); + if (strict) + { + // strict mode: next byte must be EOF + if (get_token() != token_type::end_of_input) + { + return sax->parse_error(m_lexer.get_position(), + m_lexer.get_token_string(), + parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_of_input, "value"), nullptr)); + } + } + else + { + // the caller keeps using the input: position it right after + // the value by leaving the character that terminated it + m_lexer.release_lookahead(); + } } return result; @@ -30765,6 +30952,7 @@ struct formatter // NOLINT(cert-dcl58-c #undef JSON_HAS_STATIC_RTTI #undef JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON #undef JSON_BRACE_INIT_COPY_SEMANTICS + #undef JSON_PRECISE_STREAM_POSITION #endif // #include diff --git a/single_include/nlohmann/json_fwd.hpp b/single_include/nlohmann/json_fwd.hpp index 281c05efa..af776d652 100644 --- a/single_include/nlohmann/json_fwd.hpp +++ b/single_include/nlohmann/json_fwd.hpp @@ -56,6 +56,10 @@ #define JSON_BRACE_INIT_COPY_SEMANTICS 0 #endif +#ifndef JSON_PRECISE_STREAM_POSITION + #define JSON_PRECISE_STREAM_POSITION 0 +#endif + #if JSON_DIAGNOSTICS #define NLOHMANN_JSON_ABI_TAG_DIAGNOSTICS _diag #else @@ -80,21 +84,28 @@ #define NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS #endif +#if JSON_PRECISE_STREAM_POSITION + #define NLOHMANN_JSON_ABI_TAG_PRECISE_STREAM_POSITION _psp +#else + #define NLOHMANN_JSON_ABI_TAG_PRECISE_STREAM_POSITION +#endif + #ifndef NLOHMANN_JSON_NAMESPACE_NO_VERSION #define NLOHMANN_JSON_NAMESPACE_NO_VERSION 0 #endif // Construct the namespace ABI tags component -#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d) json_abi ## a ## b ## c ## d -#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d) \ - NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d) +#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e) json_abi ## a ## b ## c ## d ## e +#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d, e) \ + NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e) #define NLOHMANN_JSON_ABI_TAGS \ NLOHMANN_JSON_ABI_TAGS_CONCAT( \ NLOHMANN_JSON_ABI_TAG_DIAGNOSTICS, \ NLOHMANN_JSON_ABI_TAG_LEGACY_DISCARDED_VALUE_COMPARISON, \ NLOHMANN_JSON_ABI_TAG_DIAGNOSTIC_POSITIONS, \ - NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS) + NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS, \ + NLOHMANN_JSON_ABI_TAG_PRECISE_STREAM_POSITION) // Construct the namespace version component #define NLOHMANN_JSON_NAMESPACE_VERSION_CONCAT_EX(major, minor, patch) \ diff --git a/tests/abi/config/default.cpp b/tests/abi/config/default.cpp index d0b4ba54b..8f66dfbf1 100644 --- a/tests/abi/config/default.cpp +++ b/tests/abi/config/default.cpp @@ -36,6 +36,10 @@ TEST_CASE("default namespace") expected += "_bics"; #endif +#if JSON_PRECISE_STREAM_POSITION + expected += "_psp"; +#endif + expected += "_v" STRINGIZE(NLOHMANN_JSON_VERSION_MAJOR); expected += "_" STRINGIZE(NLOHMANN_JSON_VERSION_MINOR); expected += "_" STRINGIZE(NLOHMANN_JSON_VERSION_PATCH) "::basic_json"; diff --git a/tests/abi/config/noversion.cpp b/tests/abi/config/noversion.cpp index 789107181..12b7603b9 100644 --- a/tests/abi/config/noversion.cpp +++ b/tests/abi/config/noversion.cpp @@ -37,6 +37,10 @@ TEST_CASE("default namespace without version component") expected += "_bics"; #endif +#if JSON_PRECISE_STREAM_POSITION + expected += "_psp"; +#endif + expected += "::basic_json"; // fallback for Clang diff --git a/tests/src/unit-deserialization.cpp b/tests/src/unit-deserialization.cpp index 0dbfdd15c..28359804b 100644 --- a/tests/src/unit-deserialization.cpp +++ b/tests/src/unit-deserialization.cpp @@ -25,6 +25,7 @@ using nlohmann::json; #include #include #include +#include #include #if defined(_WIN32) @@ -1233,6 +1234,57 @@ TEST_CASE("deserialization") } } + SECTION("stream position after extraction without JSON_PRECISE_STREAM_POSITION (#5340)") + { + // By default, the character that terminates a number is consumed, so + // the stream is left one byte too far after a number (and only after a + // number). JSON_PRECISE_STREAM_POSITION changes this; see + // unit-precise-stream-position.cpp. These checks pin the default. + const auto remaining = [](std::istream & is) + { + return std::string(std::istreambuf_iterator(is), std::istreambuf_iterator()); + }; + + SECTION("the character after a number is consumed") + { + std::istringstream ss("1true"); + json j; + ss >> j; + CHECK(j == 1); + CHECK(remaining(ss) == "rue"); + } + + SECTION("the character after other values is not consumed") + { + std::istringstream ss("[1]true"); + json j; + ss >> j; + CHECK(j == json::parse("[1]")); + CHECK(remaining(ss) == "true"); + } + + SECTION("comma-separated numbers can be read one by one") + { + std::istringstream ss("1,2,3"); + json j1, j2, j3; + ss >> j1 >> j2 >> j3; + CHECK(j1 == 1); + CHECK(j2 == 2); + CHECK(j3 == 3); + } + + SECTION("std::getline after a number skips the line break") + { + std::istringstream ss("42\nfoo"); + json j; + std::string line; + ss >> j; + std::getline(ss, line); + CHECK(j == 42); + CHECK(line == "foo"); + } + } + // build with C++20 // JSON_HAS_CPP_20 #if defined(__cpp_char8_t) diff --git a/tests/src/unit-precise-stream-position.cpp b/tests/src/unit-precise-stream-position.cpp new file mode 100644 index 000000000..5b6bff682 --- /dev/null +++ b/tests/src/unit-precise-stream-position.cpp @@ -0,0 +1,237 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#include "doctest_compatibility.h" + +// This file tests the opt-in JSON_PRECISE_STREAM_POSITION, so it defines the +// macro itself rather than relying on a -D flag, and runs in every build. The +// default behavior is pinned in unit-deserialization.cpp. +#ifdef JSON_PRECISE_STREAM_POSITION + #undef JSON_PRECISE_STREAM_POSITION +#endif + +#define JSON_PRECISE_STREAM_POSITION 1 + +#include +using nlohmann::json; + +#include +#include +#include +#include +#include +#include + +#define STRINGIZE_EX(x) #x +#define STRINGIZE(x) STRINGIZE_EX(x) + +namespace +{ +// A streambuf that keeps no get area at all and refuses every putback: with an +// empty get area, sungetc() always ends up in pbackfail(). Used to check that +// the character terminating a number is left in the input without relying on +// the streambuf being able to put a consumed character back. +class no_putback_streambuf : public std::streambuf +{ + public: + explicit no_putback_streambuf(std::string s) : m_data(std::move(s)) {} + + protected: + // peek at the next character without consuming it + int_type underflow() override + { + if (m_pos >= m_data.size()) + { + return traits_type::eof(); + } + return traits_type::to_int_type(m_data[m_pos]); + } + + // consume the next character + int_type uflow() override + { + if (m_pos >= m_data.size()) + { + return traits_type::eof(); + } + return traits_type::to_int_type(m_data[m_pos++]); + } + + int_type pbackfail(int_type /*c*/) override + { + return traits_type::eof(); + } + + private: + std::string m_data; + std::size_t m_pos = 0; +}; + +// read the characters that are left in a stream +std::string remaining(std::istream& is) +{ + std::string result; + char c = 0; + while (is.get(c)) + { + result += c; + } + return result; +} +} // namespace + +TEST_CASE("JSON_PRECISE_STREAM_POSITION") +{ + SECTION("the macro is part of the ABI tag") + { + const std::string ns = STRINGIZE(NLOHMANN_JSON_NAMESPACE); + // other tags may come before it, e.g. json_abi_diag_psp + CHECK(ns.find("_psp") != std::string::npos); + } + + SECTION("a number does not consume the character that terminates it") + { + // a number is only terminated by the character following it; that + // character must be given back so the stream is positioned right + // after the value + const std::vector> tests = + { + {"1true", "true"}, + {"1[2]", "[2]"}, + {"1{}", "{}"}, + {R"(1"a")", R"("a")"}, + {"1 true", " true"}, + {"12,", ","}, + {"-0.5e3x", "x"}, + {"1null", "null"} + }; + + for (const auto& test : tests) + { + CAPTURE(test.first); + std::istringstream ss(test.first); + json j; + ss >> j; + CHECK(j == json::parse(test.first.substr(0, test.first.size() - test.second.size()))); + CHECK(remaining(ss) == test.second); + } + } + + SECTION("values that are self-delimiting are unaffected") + { + const std::vector> tests = + { + {"truefalse", "false"}, + {"[1][2]", "[2]"}, + {R"({"a":1}{"b":2})", R"({"b":2})"}, + {R"("a""b")", R"("b")"}, + {"null null", " null"} + }; + + for (const auto& test : tests) + { + CAPTURE(test.first); + std::istringstream ss(test.first); + json j; + ss >> j; + CHECK(remaining(ss) == test.second); + } + } + + SECTION("a number at the end of the input leaves nothing behind") + { + for (const std::string s : + {"1", "12", "-3.5e2", " 7 " + }) + { + CAPTURE(s); + std::istringstream ss(s); + json j; + ss >> j; + CHECK(remaining(ss).find_first_not_of(" \t\n\r") == std::string::npos); + } + } + + SECTION("repeated extraction of concatenated values") + { + std::istringstream ss(R"(1true[2]3"x"{"a":4}5)"); + const std::vector expected = + { + json(1), json(true), json::parse("[2]"), json(3), + json("x"), json::parse(R"({"a":4})"), json(5) + }; + + for (const auto& e : expected) + { + json j; + ss >> j; + CHECK(j == e); + } + } + + SECTION("differences to the default behavior") + { + // both of these work by accident without the macro, because the + // character after a number is swallowed; see unit-deserialization.cpp + + SECTION("a separator after a number is not skipped") + { + std::istringstream ss("1,2"); + json j; + ss >> j; + CHECK(j == 1); + CHECK_THROWS_AS(ss >> j, json::parse_error&); + } + + SECTION("std::getline after a number sees the line break") + { + std::istringstream ss("42\nfoo"); + json j; + std::string line; + ss >> j; + std::getline(ss, line); + CHECK(j == 42); + CHECK(line.empty()); + std::getline(ss, line); + CHECK(line == "foo"); + } + } + + SECTION("sax_parse with strict == false") + { + std::istringstream ss("1true"); + json j; + nlohmann::detail::json_sax_dom_parser sdp(j, true); + CHECK(json::sax_parse(ss, &sdp, nlohmann::detail::input_format_t::json, false)); + CHECK(j == 1); + CHECK(remaining(ss) == "true"); + } + + SECTION("strict parsing still rejects trailing data") + { + std::istringstream ss("1true"); + json _; + CHECK_THROWS_WITH_AS(_ = json::parse(ss), + "[json.exception.parse_error.101] parse error at line 1, column 5: syntax error while parsing value - unexpected true literal; expected end of input", json::parse_error&); + + std::istringstream ss2("1true"); + CHECK_FALSE(json::accept(ss2)); + } + + SECTION("a streambuf that cannot put back is not needed") + { + // the terminating character is never consumed, so no putback + // position is required + no_putback_streambuf buf("1true"); + std::istream is(&buf); + json j; + is >> j; + CHECK(j == json(1)); + CHECK(remaining(is) == "true"); + } +} diff --git a/tools/generate_natvis/generate_natvis.py b/tools/generate_natvis/generate_natvis.py index 968690abe..fb1210db1 100755 --- a/tools/generate_natvis/generate_natvis.py +++ b/tools/generate_natvis/generate_natvis.py @@ -20,7 +20,7 @@ if __name__ == '__main__': namespaces = ['nlohmann'] abi_prefix = 'json_abi' - abi_tags = ['_diag', '_ldvcmp', '_dp', '_bics'] + abi_tags = ['_diag', '_ldvcmp', '_dp', '_bics', '_psp'] version = '_v' + args.version.replace('.', '_') inline_namespaces = [] From 02dd3e67f2ec25539d9763a7137a1950db7e94d4 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 25 Sep 2026 17:58:59 +0200 Subject: [PATCH 43/64] Fix stack overflow and exponential runtime when comparing nested values (#5390) * Compare values without recursing, and without comparing them twice Comparing two values compared their containers, which compare their elements, which brought the comparison back once per nesting level. Two values nested deeply enough exhausted the call stack and terminated the process with a segmentation fault - the same bug as #5387, in the last operation that still had it. Worse, an ordered comparison took exponentially long in the nesting depth before C++20. std::vector's operator< is a lexicographical comparison, which asks whether an element is less than its counterpart and then whether the counterpart is less than it - two full comparisons of everything below that element, at every level. Comparing two equal values nested 30 levels deep, which is nothing unusual, took 3.8 seconds; 40 levels would have taken an hour, and nothing about the value has to be pathological to get there. C++20 is unaffected: std::lexicographical_compare_three_way asks once. Compare a value that is nested too deeply to descend into on an explicit stack instead, in a single pass that yields less, equal, greater or unordered at once. Equality and the three-way comparison descend as they always did for the first 128 levels, which nothing measurable costs them; an ordered comparison no longer descends at all, which is what takes the exponent out of it. Objects and arrays that are not nested deeply are otherwise compared exactly as before. The results are unchanged for every pair of values: 68121 comparisons of a corpus that covers NaN, discarded values, mixed number types, binary values, empty containers and both object types are identical to develop, in C++11, C++17 and C++20, with and without thread_local storage and legacy discarded comparison. Reproducing that meant reproducing two subtleties: a lexicographic comparison steps over a pair it cannot order, where a three-way comparison stops at it, and an object compares its keys with < where its entries are ordered but with == where they are only checked for equality - not with the object's own comparator, which for nlohmann::ordered_map tells equality. Equality needs no ordering, so it no longer asks for any: a key or string type that can only be compared for equality still works. Measured (medians of 7 interleaved runs, clang -O3, C++11): comparing two equal values nested 30 levels deep 3778 ms -> 0.002 ms; ordering flat objects -33.6%; ordering flat arrays of numbers +27.3%, the one shape that pays for the single pass; equality unchanged throughout. Signed-off-by: Niels Lohmann * Describe comparison in the no-thread-local docs and CI target Comparing two values now bounds its descent with a thread_local counter just as copying does, so the JSON_NO_THREAD_LOCAL page, the macro overview and the ci_test_no_thread_local target cover both rather than copying alone. Also record what switching the macro on costs a comparison: on the benchmark documents, comparing two equal values takes 10% to 90% longer. Signed-off-by: Niels Lohmann * Take the descent flag as an argument rather than testing it MSVC reports the test of a constant as C4127 ("conditional expression is constant"), which the Windows builds treat as an error: may_descend is false for operator<, so the operand short-circuits the whole condition. Passing it to compare_descent_exhausted() puts the test where the value is an ordinary parameter, and leaves the call sites with no condition of their own. Signed-off-by: Niels Lohmann * Note the comparison fallback in the no-thread-local documentation The macro page describes what the library defines JSON_NO_THREAD_LOCAL for by itself in terms of copying alone; comparing falls back the same way. Signed-off-by: Niels Lohmann * Parenthesise the reserve() computation in the comparison test clang-tidy reports the mixed * and + as readability-math-missing- parentheses, as it does for the identical line in the copy test. Signed-off-by: Niels Lohmann * Use the shared descent bookkeeping rather than a second set Comparing kept a thread_local count, a limit and a guard of its own beside the ones copying already had, all three the same thing under a different name. They are gone; the shared count, limit and guard do the work. The guard grows a second constructor here, because the comparison operators are written as a macro and a macro cannot use the preprocessor: it cannot look the count up behind an #ifdef the way copy_structured does, so the guard looks it up for it. nesting_depth_exhausted() arrives for the same reason - whether an operator descends at all is a constant at every call site, and testing it there is what MSVC reports as C4127. Also say in compare_leaves what happens to a pair that is an array on one side and an object on the other, since the answer is not obvious from the code: an operator only descends into two values of the same type, so such a pair is told apart by its types alone - unequal, and ordered the way the types are - exactly as it is above the bound. And record what the explicit stack costs: the comparison operators are noexcept and the container comparison this replaces allocated nothing, so running out of memory here ends the process instead of throwing. It takes a value nested past the bound and an exhausted heap to reach, and the same comparison used to exhaust the call stack, but it is a new way to fail. Signed-off-by: Niels Lohmann * Amalgamate Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- cmake/ci.cmake | 8 +- .../docs/api/macros/json_no_thread_local.md | 19 +- docs/mkdocs/docs/features/macros.md | 5 +- include/nlohmann/json.hpp | 304 +++++++++++++++++- single_include/nlohmann/json.hpp | 304 +++++++++++++++++- tests/src/unit-large_json.cpp | 48 +++ 6 files changed, 659 insertions(+), 29 deletions(-) diff --git a/cmake/ci.cmake b/cmake/ci.cmake index a99788633..752bdc6f8 100644 --- a/cmake/ci.cmake +++ b/cmake/ci.cmake @@ -300,10 +300,10 @@ add_custom_target(ci_test_skiplibraryversioncheck # Disable thread-local storage. ############################################################################### -# Without thread-local storage, the copy constructor cannot bound its descent -# and copies every object and array without the call stack. That path is -# otherwise only reached by values nested deeper than the bound, so this target -# is what runs the whole test suite through it. +# Without thread-local storage, copying and comparing cannot bound their +# descent and handle every object and array without the call stack. Those paths +# are otherwise only reached by values nested deeper than the bound, so this +# target is what runs the whole test suite through them. add_custom_target(ci_test_no_thread_local COMMAND ${CMAKE_COMMAND} -DCMAKE_BUILD_TYPE=Debug -GNinja diff --git a/docs/mkdocs/docs/api/macros/json_no_thread_local.md b/docs/mkdocs/docs/api/macros/json_no_thread_local.md index 126116ec3..0a001ac36 100644 --- a/docs/mkdocs/docs/api/macros/json_no_thread_local.md +++ b/docs/mkdocs/docs/api/macros/json_no_thread_local.md @@ -7,16 +7,16 @@ When defined, the library does not use `#!cpp thread_local` storage. This is relevant for the few environments whose toolchain does not support it. -The copy constructor copies the first levels of a value by copying the containers, which copy their elements, and -completes whatever is nested deeper than that without the call stack, so that copying a value cannot exhaust the stack -however deeply it is nested. It counts the levels it has descended into in a `#!cpp thread_local` variable, as a counter -shared between threads would be raced. +Copying a value and comparing two values both descend into the first levels by letting the containers copy or compare +themselves, and finish whatever is nested deeper than that without the call stack, so that neither can exhaust the stack +however deeply the values are nested. Each counts the levels it has descended into in a `#!cpp thread_local` variable, as +a counter shared between threads would be raced. -Without that counter, no descent can be bounded safely, so objects and arrays are copied without the call stack right -away. Copying keeps working exactly as it does otherwise - the same values come out, and deeply nested values are copied -just as safely - but copying is slower, because the containers no longer copy themselves. Copying the benchmark -documents takes 9% (`canada.json`) to 34% (`twitter.json`) longer; values built mostly from objects are affected the -most. +Without those counters, no descent can be bounded safely, so objects and arrays are copied and compared without the call +stack right away. Both keep working exactly as they do otherwise - the same values come out, the same comparisons hold, +and deeply nested values are handled just as safely - but both are slower, because the containers no longer copy or +compare themselves. Copying the benchmark documents takes 9% (`canada.json`) to 34% (`twitter.json`) longer, and +comparing two equal ones 10% (`citm_catalog.json`) to 90% (`canada.json`) longer. ## Default definition @@ -28,6 +28,7 @@ By default, `#!cpp JSON_NO_THREAD_LOCAL` is not defined. The library defines it by itself for Clang targeting MinGW, which does not survive the `#!cpp thread_local` storage: copying a value segfaults there, with both old and current Clang versions, while GCC targeting MinGW is unaffected. +Copying and comparing fall back to working without the call stack there, as they do whenever the macro is defined. ## Examples diff --git a/docs/mkdocs/docs/features/macros.md b/docs/mkdocs/docs/features/macros.md index 7d6d23148..2bc8a4a5b 100644 --- a/docs/mkdocs/docs/features/macros.md +++ b/docs/mkdocs/docs/features/macros.md @@ -93,8 +93,9 @@ See [full documentation of `JSON_NO_IO`](../api/macros/json_no_io.md). ## `JSON_NO_THREAD_LOCAL` -When defined, the library does not use `#!cpp thread_local` storage. Copying a value then always avoids the call stack -rather than descending into a bounded number of levels first, which is slower but yields the same values. +When defined, the library does not use `#!cpp thread_local` storage. Copying a value and comparing two values then +always avoid the call stack rather than descending into a bounded number of levels first, which is slower but yields the +same values and the same comparisons. See [full documentation of `JSON_NO_THREAD_LOCAL`](../api/macros/json_no_thread_local.md). diff --git a/include/nlohmann/json.hpp b/include/nlohmann/json.hpp index 09dca4694..0123bad29 100644 --- a/include/nlohmann/json.hpp +++ b/include/nlohmann/json.hpp @@ -924,6 +924,31 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } #endif + /*! + @brief whether a descent must stop here and finish without the call stack + + @a may_descend says whether the operator descends at all; it is a constant + at every call site, and is passed rather than tested by the caller so that + the test does not become a constant condition there, which MSVC reports as + C4127. + + The comparison operators use this rather than @ref nesting_depth_guard::okay, + because they are written as a macro and a macro cannot use the preprocessor + the way the guard's constructor does; @ref copy_structured, which can, asks + the guard instead and never calls this. + */ + static bool nesting_depth_exhausted(bool may_descend = true) noexcept + { +#ifdef JSON_NO_THREAD_LOCAL + // without a count of its own per thread, a descent cannot be bounded + // without racing another one, so none is made + static_cast(may_descend); + return true; +#else + return !may_descend || nesting_depth() >= nesting_depth_limit(); +#endif + } + /*! @brief counts one level of a bounded descent for as long as it runs, and reports whether the descent was still within the limit when it began @@ -1243,6 +1268,253 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } + /// the result of comparing two values, including values that cannot be + /// ordered at all, such as a discarded value or a NaN + enum class compare_result { less, equal, greater, unordered }; + +#if JSON_HAS_THREE_WAY_COMPARISON + /// @brief the ordering that @a result stands for + static std::partial_ordering to_partial_ordering(compare_result result) noexcept // *NOPAD* + { + switch (result) + { + case compare_result::less: + return std::partial_ordering::less; + case compare_result::greater: + return std::partial_ordering::greater; + case compare_result::equal: + return std::partial_ordering::equivalent; + case compare_result::unordered: + default: + return std::partial_ordering::unordered; + } + } +#endif + + /*! + @brief compare two values that are not both an array or both an object + + Such a pair is compared by the operators themselves, which cannot descend + into it and therefore cannot recurse. + + That holds for a pair whose types differ as much as for a pair of leaves: an + array and an object are told apart by their types alone, because an operator + only ever descends into two values of the same type. So `==` reports them as + unequal without looking inside either, and an ordering falls back to the + order of the types - an object sorts before an array - exactly as it does + for a value that is not nested deeply enough to get here. + */ + template + static compare_result compare_leaves(const_reference lhs, const_reference rhs) noexcept + { + if (lhs == rhs) + { + return compare_result::equal; + } + + return order_leaves(lhs, rhs, std::integral_constant {}); + } + + /*! + @brief compare two object keys + + An object compares its entries as pairs of a key and a value, so its keys + are compared exactly as std::pair compares them: with < where the objects + are being ordered, and with == where they are only checked for equality. + Note that this is not the object's own comparator, which for a vector-backed + object type such as nlohmann::ordered_map tells equality rather than order. + */ + static compare_result compare_keys(const typename object_t::key_type& lhs, + const typename object_t::key_type& rhs, + std::true_type /*ordered*/) + { + if (lhs < rhs) + { + return compare_result::less; + } + + if (rhs < lhs) + { + return compare_result::greater; + } + + return compare_result::equal; + } + + /// @brief check two object keys for equality + static compare_result compare_keys(const typename object_t::key_type& lhs, + const typename object_t::key_type& rhs, + std::false_type /*ordered*/) + { + return lhs == rhs ? compare_result::equal : compare_result::unordered; + } + + /// @brief tell apart two values that are not equal + /// @note only instantiated where the values are being ordered, as a key or + /// string type is not required to be ordered to be compared for equality + static compare_result order_leaves(const_reference lhs, const_reference rhs, std::true_type /*ordered*/) noexcept + { + if (lhs < rhs) + { + return compare_result::less; + } + + if (rhs < lhs) + { + return compare_result::greater; + } + + return compare_result::unordered; + } + + /// @brief report two values as not equal without ordering them + static compare_result order_leaves(const_reference /*lhs*/, const_reference /*rhs*/, std::false_type /*ordered*/) noexcept + { + return compare_result::unordered; + } + + /*! + @brief compare @a lhs and @a rhs without descending into them + + Reached once a comparison has descended @ref nesting_depth_limit levels, so + that comparing values cannot exhaust the call stack however deeply they are + nested. The two values are walked in lockstep on an explicit stack and + compared lexicographically, element by element in the order the containers + enumerate them - which is how the container types this library ships compare + themselves: a std::map enumerates its entries in key order, and + nlohmann::ordered_map in insertion order. An object type that enumerates its + entries in an unspecified order, such as std::unordered_map, compares them + pairwise instead; the difference could only ever show below the bound. + + Note that the stack this walks with is allocated, while the comparison + operators are noexcept and the container comparison this replaces allocated + nothing. Failing that allocation therefore ends the process rather than + throwing. It only arises for values nested past the bound, and only when + memory has run out - where the same comparison used to exhaust the call + stack instead - but it is a way to fail that the operators did not have. + */ + template + static compare_result compare_iteratively(const_reference lhs, const_reference rhs, + const bool unordered_compares_equal) noexcept + { + /// a pair of containers being compared in lockstep + struct frame + { + const basic_json* lhs_value{nullptr}; + const basic_json* rhs_value{nullptr}; + typename array_t::const_iterator lhs_array_it{}; + typename array_t::const_iterator rhs_array_it{}; + typename object_t::const_iterator lhs_object_it{}; + typename object_t::const_iterator rhs_object_it{}; + }; + + std::vector stack; + const basic_json* left = &lhs; + const basic_json* right = &rhs; + + for (;;) + { + const auto type = left->m_data.m_type; + + if (type == right->m_data.m_type && (type == value_t::array || type == value_t::object)) + { + // descend: the elements decide, and are compared further down + stack.emplace_back(); + frame& pushed = stack.back(); + pushed.lhs_value = left; + pushed.rhs_value = right; + + if (type == value_t::array) + { + pushed.lhs_array_it = left->m_data.m_value.array->cbegin(); + pushed.rhs_array_it = right->m_data.m_value.array->cbegin(); + } + else + { + pushed.lhs_object_it = left->m_data.m_value.object->cbegin(); + pushed.rhs_object_it = right->m_data.m_value.object->cbegin(); + } + } + else + { + const compare_result result = compare_leaves(*left, *right); + + // Values that cannot be ordered - a NaN, say - end an ordered + // comparison for std::lexicographical_compare_three_way, but + // std::lexicographical_compare treats them as equivalent and + // carries on with the next element. Both are reproduced here, + // so that a value nested too deeply to descend into compares + // exactly as one that is not. + if (result != compare_result::equal && + !(unordered_compares_equal && result == compare_result::unordered)) + { + return result; + } + } + + // walk back up past the containers that are exhausted, then take the + // next pair of elements from the innermost one that is not + for (;;) + { + if (stack.empty()) + { + return compare_result::equal; + } + + frame& current = stack.back(); + const bool is_object = current.lhs_value->m_data.m_type == value_t::object; + + const bool lhs_done = is_object + ? current.lhs_object_it == current.lhs_value->m_data.m_value.object->cend() + : current.lhs_array_it == current.lhs_value->m_data.m_value.array->cend(); + const bool rhs_done = is_object + ? current.rhs_object_it == current.rhs_value->m_data.m_value.object->cend() + : current.rhs_array_it == current.rhs_value->m_data.m_value.array->cend(); + + if (lhs_done || rhs_done) + { + // whichever ran out first holds the smaller container; if + // both did, they are equal and the container above decides + if (lhs_done != rhs_done) + { + return lhs_done ? compare_result::less : compare_result::greater; + } + + stack.pop_back(); + continue; + } + + if (is_object) + { + // an entry is a key and a value, and the key decides first + const compare_result key_result = + compare_keys(current.lhs_object_it->first, current.rhs_object_it->first, + std::integral_constant {}); + + if (key_result != compare_result::equal) + { + return key_result; + } + + left = &(current.lhs_object_it->second); + right = &(current.rhs_object_it->second); + ++current.lhs_object_it; + ++current.rhs_object_it; + } + else + { + left = &(*current.lhs_array_it); + right = &(*current.rhs_array_it); + ++current.lhs_array_it; + ++current.rhs_array_it; + } + + break; + } + } + } + + /// @brief restore the parent pointers after erasing from an object /// ordered_json keeps its members in a vector, and erasing a member /// re-constructs every member after it in place, which resets their @@ -4183,7 +4455,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // because any negative signed value is smaller than any unsigned value. // Otherwise, the non-negative signed value is cast to unsigned before the // comparison to avoid wraparound. -#define JSON_IMPLEMENT_OPERATOR(op, null_result, unordered_result, default_result) \ +#define JSON_IMPLEMENT_OPERATOR(op, null_result, unordered_result, default_result, deep_result, may_descend) \ const auto lhs_type = lhs.type(); \ const auto rhs_type = rhs.type(); \ \ @@ -4192,11 +4464,25 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec switch (lhs_type) \ { \ case value_t::array: \ + { \ + if (JSON_HEDLEY_UNLIKELY(nesting_depth_exhausted(may_descend))) \ + { \ + return (deep_result); \ + } \ + const nesting_depth_guard guard; \ return (*lhs.m_data.m_value.array) op (*rhs.m_data.m_value.array); \ - \ + } \ + \ case value_t::object: \ + { \ + if (JSON_HEDLEY_UNLIKELY(nesting_depth_exhausted(may_descend))) \ + { \ + return (deep_result); \ + } \ + const nesting_depth_guard guard; \ return (*lhs.m_data.m_value.object) op (*rhs.m_data.m_value.object); \ - \ + } \ + \ case value_t::null: \ return (null_result); \ \ @@ -4296,7 +4582,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_PRAGMA(GCC diagnostic ignored "-Wfloat-equal") #endif const_reference lhs = *this; - JSON_IMPLEMENT_OPERATOR( ==, true, false, false) + JSON_IMPLEMENT_OPERATOR( ==, true, false, false, + compare_iteratively(lhs, rhs, false) == compare_result::equal, true) #ifdef __GNUC__ JSON_HEDLEY_DIAGNOSTIC_POP #endif @@ -4321,7 +4608,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_IMPLEMENT_OPERATOR(<=>, // *NOPAD* std::partial_ordering::equivalent, std::partial_ordering::unordered, - lhs_type <=> rhs_type) // *NOPAD* + lhs_type <=> rhs_type, // *NOPAD* + to_partial_ordering(compare_iteratively(lhs, rhs, false)), true) } /// @brief comparison: 3-way @@ -4388,7 +4676,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_DIAGNOSTIC_PUSH JSON_HEDLEY_PRAGMA(GCC diagnostic ignored "-Wfloat-equal") #endif - JSON_IMPLEMENT_OPERATOR( ==, true, false, false) + JSON_IMPLEMENT_OPERATOR( ==, true, false, false, + compare_iteratively(lhs, rhs, false) == compare_result::equal, true) #ifdef __GNUC__ JSON_HEDLEY_DIAGNOSTIC_POP #endif @@ -4444,7 +4733,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // default_result is used if we cannot compare values. In that case, // we compare types. Note we have to call the operator explicitly, // because MSVC has problems otherwise. - JSON_IMPLEMENT_OPERATOR( <, false, false, operator<(lhs_type, rhs_type)) + JSON_IMPLEMENT_OPERATOR( <, false, false, operator<(lhs_type, rhs_type), + compare_iteratively(lhs, rhs, true) == compare_result::less, false) } /// @brief comparison: less than diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index e3ca0f0d0..4567a2b32 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -25569,6 +25569,31 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } #endif + /*! + @brief whether a descent must stop here and finish without the call stack + + @a may_descend says whether the operator descends at all; it is a constant + at every call site, and is passed rather than tested by the caller so that + the test does not become a constant condition there, which MSVC reports as + C4127. + + The comparison operators use this rather than @ref nesting_depth_guard::okay, + because they are written as a macro and a macro cannot use the preprocessor + the way the guard's constructor does; @ref copy_structured, which can, asks + the guard instead and never calls this. + */ + static bool nesting_depth_exhausted(bool may_descend = true) noexcept + { +#ifdef JSON_NO_THREAD_LOCAL + // without a count of its own per thread, a descent cannot be bounded + // without racing another one, so none is made + static_cast(may_descend); + return true; +#else + return !may_descend || nesting_depth() >= nesting_depth_limit(); +#endif + } + /*! @brief counts one level of a bounded descent for as long as it runs, and reports whether the descent was still within the limit when it began @@ -25888,6 +25913,253 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec } + /// the result of comparing two values, including values that cannot be + /// ordered at all, such as a discarded value or a NaN + enum class compare_result { less, equal, greater, unordered }; + +#if JSON_HAS_THREE_WAY_COMPARISON + /// @brief the ordering that @a result stands for + static std::partial_ordering to_partial_ordering(compare_result result) noexcept // *NOPAD* + { + switch (result) + { + case compare_result::less: + return std::partial_ordering::less; + case compare_result::greater: + return std::partial_ordering::greater; + case compare_result::equal: + return std::partial_ordering::equivalent; + case compare_result::unordered: + default: + return std::partial_ordering::unordered; + } + } +#endif + + /*! + @brief compare two values that are not both an array or both an object + + Such a pair is compared by the operators themselves, which cannot descend + into it and therefore cannot recurse. + + That holds for a pair whose types differ as much as for a pair of leaves: an + array and an object are told apart by their types alone, because an operator + only ever descends into two values of the same type. So `==` reports them as + unequal without looking inside either, and an ordering falls back to the + order of the types - an object sorts before an array - exactly as it does + for a value that is not nested deeply enough to get here. + */ + template + static compare_result compare_leaves(const_reference lhs, const_reference rhs) noexcept + { + if (lhs == rhs) + { + return compare_result::equal; + } + + return order_leaves(lhs, rhs, std::integral_constant {}); + } + + /*! + @brief compare two object keys + + An object compares its entries as pairs of a key and a value, so its keys + are compared exactly as std::pair compares them: with < where the objects + are being ordered, and with == where they are only checked for equality. + Note that this is not the object's own comparator, which for a vector-backed + object type such as nlohmann::ordered_map tells equality rather than order. + */ + static compare_result compare_keys(const typename object_t::key_type& lhs, + const typename object_t::key_type& rhs, + std::true_type /*ordered*/) + { + if (lhs < rhs) + { + return compare_result::less; + } + + if (rhs < lhs) + { + return compare_result::greater; + } + + return compare_result::equal; + } + + /// @brief check two object keys for equality + static compare_result compare_keys(const typename object_t::key_type& lhs, + const typename object_t::key_type& rhs, + std::false_type /*ordered*/) + { + return lhs == rhs ? compare_result::equal : compare_result::unordered; + } + + /// @brief tell apart two values that are not equal + /// @note only instantiated where the values are being ordered, as a key or + /// string type is not required to be ordered to be compared for equality + static compare_result order_leaves(const_reference lhs, const_reference rhs, std::true_type /*ordered*/) noexcept + { + if (lhs < rhs) + { + return compare_result::less; + } + + if (rhs < lhs) + { + return compare_result::greater; + } + + return compare_result::unordered; + } + + /// @brief report two values as not equal without ordering them + static compare_result order_leaves(const_reference /*lhs*/, const_reference /*rhs*/, std::false_type /*ordered*/) noexcept + { + return compare_result::unordered; + } + + /*! + @brief compare @a lhs and @a rhs without descending into them + + Reached once a comparison has descended @ref nesting_depth_limit levels, so + that comparing values cannot exhaust the call stack however deeply they are + nested. The two values are walked in lockstep on an explicit stack and + compared lexicographically, element by element in the order the containers + enumerate them - which is how the container types this library ships compare + themselves: a std::map enumerates its entries in key order, and + nlohmann::ordered_map in insertion order. An object type that enumerates its + entries in an unspecified order, such as std::unordered_map, compares them + pairwise instead; the difference could only ever show below the bound. + + Note that the stack this walks with is allocated, while the comparison + operators are noexcept and the container comparison this replaces allocated + nothing. Failing that allocation therefore ends the process rather than + throwing. It only arises for values nested past the bound, and only when + memory has run out - where the same comparison used to exhaust the call + stack instead - but it is a way to fail that the operators did not have. + */ + template + static compare_result compare_iteratively(const_reference lhs, const_reference rhs, + const bool unordered_compares_equal) noexcept + { + /// a pair of containers being compared in lockstep + struct frame + { + const basic_json* lhs_value{nullptr}; + const basic_json* rhs_value{nullptr}; + typename array_t::const_iterator lhs_array_it{}; + typename array_t::const_iterator rhs_array_it{}; + typename object_t::const_iterator lhs_object_it{}; + typename object_t::const_iterator rhs_object_it{}; + }; + + std::vector stack; + const basic_json* left = &lhs; + const basic_json* right = &rhs; + + for (;;) + { + const auto type = left->m_data.m_type; + + if (type == right->m_data.m_type && (type == value_t::array || type == value_t::object)) + { + // descend: the elements decide, and are compared further down + stack.emplace_back(); + frame& pushed = stack.back(); + pushed.lhs_value = left; + pushed.rhs_value = right; + + if (type == value_t::array) + { + pushed.lhs_array_it = left->m_data.m_value.array->cbegin(); + pushed.rhs_array_it = right->m_data.m_value.array->cbegin(); + } + else + { + pushed.lhs_object_it = left->m_data.m_value.object->cbegin(); + pushed.rhs_object_it = right->m_data.m_value.object->cbegin(); + } + } + else + { + const compare_result result = compare_leaves(*left, *right); + + // Values that cannot be ordered - a NaN, say - end an ordered + // comparison for std::lexicographical_compare_three_way, but + // std::lexicographical_compare treats them as equivalent and + // carries on with the next element. Both are reproduced here, + // so that a value nested too deeply to descend into compares + // exactly as one that is not. + if (result != compare_result::equal && + !(unordered_compares_equal && result == compare_result::unordered)) + { + return result; + } + } + + // walk back up past the containers that are exhausted, then take the + // next pair of elements from the innermost one that is not + for (;;) + { + if (stack.empty()) + { + return compare_result::equal; + } + + frame& current = stack.back(); + const bool is_object = current.lhs_value->m_data.m_type == value_t::object; + + const bool lhs_done = is_object + ? current.lhs_object_it == current.lhs_value->m_data.m_value.object->cend() + : current.lhs_array_it == current.lhs_value->m_data.m_value.array->cend(); + const bool rhs_done = is_object + ? current.rhs_object_it == current.rhs_value->m_data.m_value.object->cend() + : current.rhs_array_it == current.rhs_value->m_data.m_value.array->cend(); + + if (lhs_done || rhs_done) + { + // whichever ran out first holds the smaller container; if + // both did, they are equal and the container above decides + if (lhs_done != rhs_done) + { + return lhs_done ? compare_result::less : compare_result::greater; + } + + stack.pop_back(); + continue; + } + + if (is_object) + { + // an entry is a key and a value, and the key decides first + const compare_result key_result = + compare_keys(current.lhs_object_it->first, current.rhs_object_it->first, + std::integral_constant {}); + + if (key_result != compare_result::equal) + { + return key_result; + } + + left = &(current.lhs_object_it->second); + right = &(current.rhs_object_it->second); + ++current.lhs_object_it; + ++current.rhs_object_it; + } + else + { + left = &(*current.lhs_array_it); + right = &(*current.rhs_array_it); + ++current.lhs_array_it; + ++current.rhs_array_it; + } + + break; + } + } + } + + /// @brief restore the parent pointers after erasing from an object /// ordered_json keeps its members in a vector, and erasing a member /// re-constructs every member after it in place, which resets their @@ -28828,7 +29100,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // because any negative signed value is smaller than any unsigned value. // Otherwise, the non-negative signed value is cast to unsigned before the // comparison to avoid wraparound. -#define JSON_IMPLEMENT_OPERATOR(op, null_result, unordered_result, default_result) \ +#define JSON_IMPLEMENT_OPERATOR(op, null_result, unordered_result, default_result, deep_result, may_descend) \ const auto lhs_type = lhs.type(); \ const auto rhs_type = rhs.type(); \ \ @@ -28837,11 +29109,25 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec switch (lhs_type) \ { \ case value_t::array: \ + { \ + if (JSON_HEDLEY_UNLIKELY(nesting_depth_exhausted(may_descend))) \ + { \ + return (deep_result); \ + } \ + const nesting_depth_guard guard; \ return (*lhs.m_data.m_value.array) op (*rhs.m_data.m_value.array); \ - \ + } \ + \ case value_t::object: \ + { \ + if (JSON_HEDLEY_UNLIKELY(nesting_depth_exhausted(may_descend))) \ + { \ + return (deep_result); \ + } \ + const nesting_depth_guard guard; \ return (*lhs.m_data.m_value.object) op (*rhs.m_data.m_value.object); \ - \ + } \ + \ case value_t::null: \ return (null_result); \ \ @@ -28941,7 +29227,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_PRAGMA(GCC diagnostic ignored "-Wfloat-equal") #endif const_reference lhs = *this; - JSON_IMPLEMENT_OPERATOR( ==, true, false, false) + JSON_IMPLEMENT_OPERATOR( ==, true, false, false, + compare_iteratively(lhs, rhs, false) == compare_result::equal, true) #ifdef __GNUC__ JSON_HEDLEY_DIAGNOSTIC_POP #endif @@ -28966,7 +29253,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_IMPLEMENT_OPERATOR(<=>, // *NOPAD* std::partial_ordering::equivalent, std::partial_ordering::unordered, - lhs_type <=> rhs_type) // *NOPAD* + lhs_type <=> rhs_type, // *NOPAD* + to_partial_ordering(compare_iteratively(lhs, rhs, false)), true) } /// @brief comparison: 3-way @@ -29033,7 +29321,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec JSON_HEDLEY_DIAGNOSTIC_PUSH JSON_HEDLEY_PRAGMA(GCC diagnostic ignored "-Wfloat-equal") #endif - JSON_IMPLEMENT_OPERATOR( ==, true, false, false) + JSON_IMPLEMENT_OPERATOR( ==, true, false, false, + compare_iteratively(lhs, rhs, false) == compare_result::equal, true) #ifdef __GNUC__ JSON_HEDLEY_DIAGNOSTIC_POP #endif @@ -29089,7 +29378,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // default_result is used if we cannot compare values. In that case, // we compare types. Note we have to call the operator explicitly, // because MSVC has problems otherwise. - JSON_IMPLEMENT_OPERATOR( <, false, false, operator<(lhs_type, rhs_type)) + JSON_IMPLEMENT_OPERATOR( <, false, false, operator<(lhs_type, rhs_type), + compare_iteratively(lhs, rhs, true) == compare_result::less, false) } /// @brief comparison: less than diff --git a/tests/src/unit-large_json.cpp b/tests/src/unit-large_json.cpp index f10c8be41..3734966d5 100644 --- a/tests/src/unit-large_json.cpp +++ b/tests/src/unit-large_json.cpp @@ -157,6 +157,54 @@ TEST_CASE("tests on deeply nested JSONs") CHECK(deep_depth == depth); } + SECTION("comparing") + { + // Comparing used to descend once per level, and an ordered + // comparison used to compare every pair of elements twice, once in + // each direction, which took exponentially long in the nesting + // depth. Both are gone: these finish in milliseconds, where the + // second used to take longer than anyone would wait even for a + // value nested only a few dozen levels deep. + const std::string text = std::string(depth, '[') + '0' + std::string(depth, ']'); + const json j = json::parse(text); + const json same = json::parse(text); + const json larger = json::parse(std::string(depth, '[') + '1' + std::string(depth, ']')); + + CHECK(j == same); + CHECK_FALSE(j == larger); + CHECK(j != larger); + + CHECK(j < larger); + CHECK_FALSE(larger < j); + CHECK(larger > j); + CHECK(j <= same); + CHECK(j >= same); + + // a value that ends earlier is the smaller one + const json shorter = json::parse(std::string(depth - 1, '[') + '0' + std::string(depth - 1, ']')); + CHECK_FALSE(j == shorter); + } + + SECTION("comparing objects") + { + std::string text; + text.reserve((6 * depth) + 1); + for (std::size_t i = 0; i < depth; ++i) + { + text += "{\"a\":"; + } + text += '1'; + text.append(depth, '}'); + + const json j = json::parse(text); + const json same = json::parse(text); + + CHECK(j == same); + CHECK_FALSE(j != same); + CHECK(j <= same); + CHECK(j >= same); + } + SECTION("the copy is independent of the original") { const json j = json::parse(std::string(depth, '[') + '0' + std::string(depth, ']')); From 465407f3ce046558b712fe8f9d9db74686cf9d55 Mon Sep 17 00:00:00 2001 From: Alexander Lanin Date: Fri, 25 Sep 2026 18:02:38 +0200 Subject: [PATCH 44/64] Improve error message for const fields (#2818) * Improve error message for const fields * Reject const arguments to get_to() with a clear message Reword the static_assert, add it to the C array overload of get_to() as well, and document that v must not be const. Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann Co-authored-by: Niels Lohmann --- docs/mkdocs/docs/api/basic_json/get_to.md | 5 +++++ include/nlohmann/json.hpp | 2 ++ single_include/nlohmann/json.hpp | 2 ++ 3 files changed, 9 insertions(+) diff --git a/docs/mkdocs/docs/api/basic_json/get_to.md b/docs/mkdocs/docs/api/basic_json/get_to.md index 8334ba5cf..c50c077b0 100644 --- a/docs/mkdocs/docs/api/basic_json/get_to.md +++ b/docs/mkdocs/docs/api/basic_json/get_to.md @@ -21,6 +21,10 @@ This overload is chosen if: - `ValueType` is not `basic_json`, - `json_serializer` has a `from_json()` method of the form `void from_json(const basic_json&, ValueType&)` +`v` must not be `const`. Passing a `const` object is a compile-time error. For types such as arithmetic types, enums, +and C arrays, the error is a `static_assert` that names the problem. For other types, the overload is not viable, and +the compiler reports that no matching `get_to` was found. + ## Template parameters `ValueType` @@ -67,3 +71,4 @@ Depends on the `json_serializer::from_json()` implementation. ## Version history - Since version 3.3.0. +- Added a `static_assert` with a clear message for `const` arguments in version 3.13.0. diff --git a/include/nlohmann/json.hpp b/include/nlohmann/json.hpp index 0123bad29..1090ed105 100644 --- a/include/nlohmann/json.hpp +++ b/include/nlohmann/json.hpp @@ -2553,6 +2553,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec ValueType & get_to(ValueType& v) const noexcept(noexcept( JSONSerializer::from_json(std::declval(), v))) { + static_assert(!std::is_const::value, "get_to() cannot deserialize into a const value"); JSONSerializer::from_json(*this, v); return v; } @@ -2578,6 +2579,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec noexcept(noexcept(JSONSerializer::from_json( std::declval(), v))) { + static_assert(!std::is_const::value, "get_to() cannot deserialize into a const value"); JSONSerializer::from_json(*this, v); return v; } diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 4567a2b32..8c3a1319b 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -27198,6 +27198,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec ValueType & get_to(ValueType& v) const noexcept(noexcept( JSONSerializer::from_json(std::declval(), v))) { + static_assert(!std::is_const::value, "get_to() cannot deserialize into a const value"); JSONSerializer::from_json(*this, v); return v; } @@ -27223,6 +27224,7 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec noexcept(noexcept(JSONSerializer::from_json( std::declval(), v))) { + static_assert(!std::is_const::value, "get_to() cannot deserialize into a const value"); JSONSerializer::from_json(*this, v); return v; } From e1310ad43c0d29adbe652bb140bd699d9ccf7eda Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 25 Sep 2026 20:19:43 +0200 Subject: [PATCH 45/64] Fix CI: resolve clang-tidy findings in the stream position tests (#5578) #5344 added two lines to unit-deserialization.cpp that clang-tidy reports: modernize-return-braced-init-list for the remaining() helper and readability-isolate-declaration for "json j1, j2, j3;". Signed-off-by: Niels Lohmann --- tests/src/unit-deserialization.cpp | 8 +++++--- 1 file changed, 5 insertions(+), 3 deletions(-) diff --git a/tests/src/unit-deserialization.cpp b/tests/src/unit-deserialization.cpp index 28359804b..4202644f6 100644 --- a/tests/src/unit-deserialization.cpp +++ b/tests/src/unit-deserialization.cpp @@ -1240,9 +1240,9 @@ TEST_CASE("deserialization") // the stream is left one byte too far after a number (and only after a // number). JSON_PRECISE_STREAM_POSITION changes this; see // unit-precise-stream-position.cpp. These checks pin the default. - const auto remaining = [](std::istream & is) + const auto remaining = [](std::istream & is) -> std::string { - return std::string(std::istreambuf_iterator(is), std::istreambuf_iterator()); + return {std::istreambuf_iterator(is), std::istreambuf_iterator()}; }; SECTION("the character after a number is consumed") @@ -1266,7 +1266,9 @@ TEST_CASE("deserialization") SECTION("comma-separated numbers can be read one by one") { std::istringstream ss("1,2,3"); - json j1, j2, j3; + json j1; + json j2; + json j3; ss >> j1 >> j2 >> j3; CHECK(j1 == 1); CHECK(j2 == 2); From ad2c14b985203eaf5da1869393a4d16523b07da0 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 25 Sep 2026 20:34:58 +0200 Subject: [PATCH 46/64] Document response times, supported versions, access, secrets, and dependency policies (#5580) Answer the OpenSSF Best Practices criteria that asked for policies the project follows but had not written down: - SECURITY.md: a first response within 14 days, publishing an advisory with credit once a fix is released, and that only the latest release receives security fixes. - Governance: who has access to the project's resources, how write or admin access is granted, and how CI secrets are stored and rotated. - Quality assurance: how dependencies of the build, test, and documentation tooling are pinned, scanned, and kept free of known vulnerabilities. Also update the assurance case, since comparison no longer recurses per nesting level (#5390), and point the best practices badge and links to bestpractices.dev under the program's current name. Signed-off-by: Niels Lohmann --- .github/SECURITY.md | 17 ++++++++++--- README.md | 4 +-- docs/mkdocs/docs/community/assurance_case.md | 4 +-- docs/mkdocs/docs/community/governance.md | 25 +++++++++++++++++++ .../docs/community/quality_assurance.md | 19 ++++++++++++++ docs/mkdocs/docs/home/design_goals.md | 2 +- 6 files changed, 63 insertions(+), 8 deletions(-) diff --git a/.github/SECURITY.md b/.github/SECURITY.md index 066cf1a2e..d591bf2a8 100644 --- a/.github/SECURITY.md +++ b/.github/SECURITY.md @@ -9,12 +9,23 @@ identified a security vulnerability in this repository, please use the GitHub Se Until it is published, this draft security advisory will only be visible to the maintainers of this project. Other users and teams may be added once the advisory is created. -We will send a response indicating the next steps in handling your report. After the initial reply to your report, we -will keep you informed of the progress towards a fix and full announcement and may ask for additional information or -guidance. +We will send a first response within 14 days, indicating the next steps in handling your report. After the initial +reply to your report, we will keep you informed of the progress towards a fix and full announcement and may ask for +additional information or guidance. For vulnerabilities in third-party dependencies or modules, please report them directly to the respective maintainers. +## Disclosure and credit + +Once a fix is released, we publish the security advisory and list the fixed vulnerability in the release notes. We +credit the reporter in both, unless they ask not to be named. + +## Supported versions + +Security fixes are made on the `develop` branch and shipped with the next release. Only the latest release receives +security fixes; they are not backported to older releases. A release stops receiving security fixes when the next +release is published, so please update to the latest release to get them. + ## Unofficial packages This project does not publish an official npm package. The npm package diff --git a/README.md b/README.md index becce71a1..412d05fd4 100644 --- a/README.md +++ b/README.md @@ -17,7 +17,7 @@ [![GitHub Downloads](https://img.shields.io/github/downloads/nlohmann/json/total)](https://github.com/nlohmann/json/releases) [![GitHub Issues](https://img.shields.io/github/issues/nlohmann/json.svg)](https://github.com/nlohmann/json/issues) [![Average time to resolve an issue](https://isitmaintained.com/badge/resolution/nlohmann/json.svg)](https://isitmaintained.com/project/nlohmann/json "Average time to resolve an issue") -[![CII Best Practices](https://bestpractices.coreinfrastructure.org/projects/289/badge)](https://bestpractices.coreinfrastructure.org/projects/289) +[![OpenSSF Best Practices](https://www.bestpractices.dev/projects/289/badge)](https://www.bestpractices.dev/projects/289) [![OpenSSF Scorecard](https://api.scorecard.dev/projects/github.com/nlohmann/json/badge)](https://scorecard.dev/viewer/?uri=github.com/nlohmann/json) [![Backup Status](https://app.cloudback.it/badge/nlohmann/json)](https://cloudback.it) [![GitHub Sponsors](https://img.shields.io/badge/GitHub-Sponsors-ff69b4)](https://github.com/sponsors/nlohmann) @@ -63,7 +63,7 @@ There are myriads of [JSON](https://json.org) libraries out there, and each may - **Trivial integration**. Our whole code consists of a single header file [`json.hpp`](https://github.com/nlohmann/json/blob/develop/single_include/nlohmann/json.hpp). That's it. No library, no subproject, no dependencies, no complex build system. The class is written in vanilla C++11. All in all, everything should require no adjustment of your compiler flags or project settings. The library is also included in all popular [package managers](https://json.nlohmann.me/integration/package_managers/). -- **Serious testing**. Our code is heavily [unit-tested](https://github.com/nlohmann/json/tree/develop/tests/src) and covers [100%](https://coveralls.io/r/nlohmann/json) of the code, including all exceptional behavior. Furthermore, we checked with [Valgrind](https://valgrind.org) and the [Clang Sanitizers](https://clang.llvm.org/docs/index.html) that there are no memory leaks. [Google OSS-Fuzz](https://github.com/google/oss-fuzz/tree/master/projects/json) additionally runs fuzz tests against all parsers 24/7, effectively executing billions of tests so far. To maintain high quality, the project is following the [Core Infrastructure Initiative (CII) best practices](https://bestpractices.coreinfrastructure.org/projects/289). See the [quality assurance](https://json.nlohmann.me/community/quality_assurance) overview documentation. +- **Serious testing**. Our code is heavily [unit-tested](https://github.com/nlohmann/json/tree/develop/tests/src) and covers [100%](https://coveralls.io/r/nlohmann/json) of the code, including all exceptional behavior. Furthermore, we checked with [Valgrind](https://valgrind.org) and the [Clang Sanitizers](https://clang.llvm.org/docs/index.html) that there are no memory leaks. [Google OSS-Fuzz](https://github.com/google/oss-fuzz/tree/master/projects/json) additionally runs fuzz tests against all parsers 24/7, effectively executing billions of tests so far. To maintain high quality, the project is following the [OpenSSF Best Practices](https://www.bestpractices.dev/projects/289). See the [quality assurance](https://json.nlohmann.me/community/quality_assurance) overview documentation. Other aspects were not so important to us: diff --git a/docs/mkdocs/docs/community/assurance_case.md b/docs/mkdocs/docs/community/assurance_case.md index 87d8d6f14..10f4689c1 100644 --- a/docs/mkdocs/docs/community/assurance_case.md +++ b/docs/mkdocs/docs/community/assurance_case.md @@ -43,8 +43,8 @@ that an attacker controls, passed to [`parse`](../api/basic_json/parse.md), [`ac user code. The destructor does not recurse, so destroying a deeply nested value does not exhaust the stack. - **Bounded recursion.** The JSON parser and the binary readers keep their state in explicit stacks instead of recursing per nesting level. Operations that walk a value, such as [`dump`](../api/basic_json/dump.md), copying, - hashing, and [`merge_patch`](../api/basic_json/merge_patch.md), recurse only up to a fixed depth and continue with an - explicit stack below it. Some operations, such as comparison, [`diff`](../api/basic_json/diff.md), + comparison, hashing, and [`merge_patch`](../api/basic_json/merge_patch.md), recurse only up to a fixed depth and + continue with an explicit stack below it. Some operations, such as [`diff`](../api/basic_json/diff.md), [`flatten`](../api/basic_json/flatten.md), and the binary writers, still recurse once per nesting level; work on them is in progress. Applications that process untrusted input can limit its nesting depth with a [parser callback](../features/parsing/parser_callbacks.md). diff --git a/docs/mkdocs/docs/community/governance.md b/docs/mkdocs/docs/community/governance.md index 7da85f3fe..f4127ccea 100644 --- a/docs/mkdocs/docs/community/governance.md +++ b/docs/mkdocs/docs/community/governance.md @@ -91,6 +91,31 @@ activities include (but are not limited to): Users who continue to engage with the project and its community will often find themselves becoming more and more involved. Such users may then go on to become contributors, as described above. +## Access to project resources + +The project's resources are the [GitHub repository](https://github.com/nlohmann/json) with its settings, CI workflows +and secrets, and the documentation at [json.nlohmann.me](https://json.nlohmann.me), which is built and deployed from +the repository. Currently, the project lead is the only person with write or admin access to them. + +### Granting access + +Write or admin access is only granted by the project lead, and only to a contributor whose track record in the project +the project lead has reviewed first. The role is assigned manually and is the lowest one that is needed for the task. +Access is removed when it is no longer needed. GitHub requires two-factor authentication for everyone who can modify the +repository. + +### Secrets + +The CI workflows mostly use the token that GitHub creates for each workflow run. It is read-only by default, and each +workflow requests only the additional permissions it needs. The few other credentials, such as the token for +[Semgrep](https://semgrep.dev), are stored as encrypted GitHub Actions secrets: + +- Only people with admin access can create, change, or delete them. Their values cannot be read back, not even by + admins. +- They are not passed to workflows that run for pull requests from forks. +- They must never be committed to the repository or printed in logs. +- They are rotated whenever someone with admin access leaves the project, and immediately if a leak is suspected. + ## Support All participants in the community are encouraged to provide support for new users within the project management diff --git a/docs/mkdocs/docs/community/quality_assurance.md b/docs/mkdocs/docs/community/quality_assurance.md index bd35516b8..f2404c77b 100644 --- a/docs/mkdocs/docs/community/quality_assurance.md +++ b/docs/mkdocs/docs/community/quality_assurance.md @@ -200,6 +200,25 @@ Note: Some modern features (like C++20 ranges or filesystem support) may be disa - [x] The test suite is executed with [Sanitizers](https://github.com/google/sanitizers) (address sanitizer, undefined behavior sanitizer, integer overflow detection, nullability violations). +## Dependencies + +!!! success "Requirement: No vulnerable dependencies" + + The library has no dependencies besides the C++ standard library. The tools used to build, test, and document it + are kept free of known vulnerabilities. + +- [x] GitHub Actions are pinned to a commit hash, and the Python packages used by the documentation and the tools are + pinned to exact versions. +- [x] [Dependabot](https://docs.github.com/en/code-security/dependabot) checks these dependencies daily and proposes + updates as pull requests. +- [x] Every pull request is checked with the + [dependency review action](https://github.com/actions/dependency-review-action). A pull request that adds a + dependency with a known vulnerability of any severity fails this check and is not merged. +- [x] Vulnerability alerts for dependencies are fixed or dismissed with a documented reason before the next release. + No release is made while such an alert is open. +- [x] Third-party code included in the repository for testing, such as [doctest](https://github.com/doctest/doctest), + is updated manually. + ## Style check !!! success "Requirement: Common code style" diff --git a/docs/mkdocs/docs/home/design_goals.md b/docs/mkdocs/docs/home/design_goals.md index 0a0f77029..549ba9a1a 100644 --- a/docs/mkdocs/docs/home/design_goals.md +++ b/docs/mkdocs/docs/home/design_goals.md @@ -6,7 +6,7 @@ There are myriads of [JSON](https://json.org) libraries out there, and each may - **Trivial integration**. Our whole code consists of a single header file [`json.hpp`](https://github.com/nlohmann/json/blob/develop/single_include/nlohmann/json.hpp). That's it. No library, no subproject, no dependencies, no complex build system. The class is written in vanilla C++11. All in all, everything should require no adjustment of your compiler flags or project settings. -- **Serious testing**. Our class is heavily [unit-tested](https://github.com/nlohmann/json/tree/develop/tests/src) and covers [100%](https://coveralls.io/r/nlohmann/json) of the code, including all exceptional behavior. Furthermore, we checked with [Valgrind](http://valgrind.org) and the [Clang Sanitizers](https://clang.llvm.org/docs/index.html) that there are no memory leaks. [Google OSS-Fuzz](https://github.com/google/oss-fuzz/tree/master/projects/json) additionally runs fuzz tests against all parsers 24/7, effectively executing billions of tests so far. To maintain high quality, the project is following the [Core Infrastructure Initiative (CII) best practices](https://bestpractices.coreinfrastructure.org/projects/289). +- **Serious testing**. Our class is heavily [unit-tested](https://github.com/nlohmann/json/tree/develop/tests/src) and covers [100%](https://coveralls.io/r/nlohmann/json) of the code, including all exceptional behavior. Furthermore, we checked with [Valgrind](http://valgrind.org) and the [Clang Sanitizers](https://clang.llvm.org/docs/index.html) that there are no memory leaks. [Google OSS-Fuzz](https://github.com/google/oss-fuzz/tree/master/projects/json) additionally runs fuzz tests against all parsers 24/7, effectively executing billions of tests so far. To maintain high quality, the project is following the [OpenSSF Best Practices](https://www.bestpractices.dev/projects/289). Other aspects were not so important to us: From a9ab2a62ba16540e5ae7460f05640e892597635b Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 25 Sep 2026 20:40:28 +0200 Subject: [PATCH 47/64] Cancel superseded runs of the remaining workflows (#5579) Ubuntu, Windows, macOS, and CodeQL already cancel an older run of the same workflow on the same ref. Check amalgamation, CIFuzz, Dependency Review, Flawfinder, Semgrep, Scorecard, and the labeler did not, so every push to a pull request left their earlier runs going. Give them the same concurrency group. The labeler runs on pull_request_target, where github.ref is the base branch, so it groups by pull request number. Signed-off-by: Niels Lohmann --- .github/workflows/check_amalgamation.yml | 4 ++++ .github/workflows/cifuzz.yml | 4 ++++ .github/workflows/dependency-review.yml | 4 ++++ .github/workflows/flawfinder.yml | 4 ++++ .github/workflows/labeler.yml | 6 ++++++ .github/workflows/scorecards.yml | 4 ++++ .github/workflows/semgrep.yml | 4 ++++ 7 files changed, 30 insertions(+) diff --git a/.github/workflows/check_amalgamation.yml b/.github/workflows/check_amalgamation.yml index f692e434a..670269cbb 100644 --- a/.github/workflows/check_amalgamation.yml +++ b/.github/workflows/check_amalgamation.yml @@ -3,6 +3,10 @@ name: "Check amalgamation" on: pull_request: +concurrency: + group: ${{ github.workflow }}-${{ github.ref || github.run_id }} + cancel-in-progress: true + permissions: contents: read diff --git a/.github/workflows/cifuzz.yml b/.github/workflows/cifuzz.yml index 2e50fee9c..9398fdbaf 100644 --- a/.github/workflows/cifuzz.yml +++ b/.github/workflows/cifuzz.yml @@ -1,6 +1,10 @@ name: CIFuzz on: [pull_request] +concurrency: + group: ${{ github.workflow }}-${{ github.ref || github.run_id }} + cancel-in-progress: true + permissions: contents: read diff --git a/.github/workflows/dependency-review.yml b/.github/workflows/dependency-review.yml index aa09008a3..2052c1d35 100644 --- a/.github/workflows/dependency-review.yml +++ b/.github/workflows/dependency-review.yml @@ -9,6 +9,10 @@ name: 'Dependency Review' on: [pull_request] +concurrency: + group: ${{ github.workflow }}-${{ github.ref || github.run_id }} + cancel-in-progress: true + permissions: contents: read diff --git a/.github/workflows/flawfinder.yml b/.github/workflows/flawfinder.yml index 2c8befd40..753b81e0d 100644 --- a/.github/workflows/flawfinder.yml +++ b/.github/workflows/flawfinder.yml @@ -5,6 +5,10 @@ name: flawfinder +concurrency: + group: ${{ github.workflow }}-${{ github.ref || github.run_id }} + cancel-in-progress: true + permissions: contents: read diff --git a/.github/workflows/labeler.yml b/.github/workflows/labeler.yml index 3b4571b95..a93c86ae2 100644 --- a/.github/workflows/labeler.yml +++ b/.github/workflows/labeler.yml @@ -4,6 +4,12 @@ on: pull_request_target: types: [opened, synchronize] +# pull_request_target runs on the base branch, so github.ref would put all pull +# requests into one group; group by pull request number instead +concurrency: + group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.run_id }} + cancel-in-progress: true + permissions: contents: read diff --git a/.github/workflows/scorecards.yml b/.github/workflows/scorecards.yml index 113da079a..96f87526a 100644 --- a/.github/workflows/scorecards.yml +++ b/.github/workflows/scorecards.yml @@ -14,6 +14,10 @@ on: push: branches: ["develop"] +concurrency: + group: ${{ github.workflow }}-${{ github.ref || github.run_id }} + cancel-in-progress: true + permissions: contents: read diff --git a/.github/workflows/semgrep.yml b/.github/workflows/semgrep.yml index dc326db55..922c9ca86 100644 --- a/.github/workflows/semgrep.yml +++ b/.github/workflows/semgrep.yml @@ -19,6 +19,10 @@ on: schedule: - cron: '23 2 * * 4' +concurrency: + group: ${{ github.workflow }}-${{ github.ref || github.run_id }} + cancel-in-progress: true + permissions: contents: read From ec4bdc398a1a0e00ea18abd898d11ff7277ccb50 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 25 Sep 2026 20:43:34 +0200 Subject: [PATCH 48/64] Document the benchmarks, make them build and compare versions again, and pin Google Benchmark (#5556) * Document the benchmarks, and make them build and compare versions again The benchmark project hasn't configured since #4793: download_test_data.cmake compiles cmake/detect_libcpp_version.cpp relative to CMAKE_SOURCE_DIR, which is tests/benchmarks when that is the top-level project, so try_run fails and so does `make run_benchmarks`. The path is now relative to the module itself, which is the same file for the main build. The Dump benchmark discarded dump()'s result, which is [[nodiscard]] by now; it warned, and left the optimizer free to shorten the loop. The result is now kept with benchmark::DoNotOptimize. A new cache variable, JSON_BENCHMARK_INCLUDE_DIR, names the directory holding the nlohmann/json.hpp to benchmark (single_include by default, as before), so the same benchmarks can be built against two versions and compared. tests/benchmarks/README.md documents what is measured, how to build and run the benchmarks, how to read the output, and how to compare two versions with Google Benchmark's compare.py; it recommends doing so by hand before a release rather than in CI. Signed-off-by: Niels Lohmann * Point ci_benchmarks at tests/benchmarks The target has configured ${PROJECT_SOURCE_DIR}/benchmarks since it was added in #2561, but the benchmarks live in tests/benchmarks. Signed-off-by: Niels Lohmann * Pin Google Benchmark to release 1.9.5 The benchmarks fetched Google Benchmark's main branch, so two builds on different days could measure with different library code, and CMake 3.30 and later warn that the single-argument FetchContent_Populate() is deprecated. Fetch the 1.9.5 release archive, verified by its SHA-256, with FetchContent_MakeAvailable() instead. That needs CMake 3.14; Google Benchmark itself already needed 3.13. Its -Werror is switched off, so a newer compiler's new warnings cannot break the pinned release, and its install rules are no longer added. Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- cmake/ci.cmake | 2 +- cmake/download_test_data.cmake | 2 +- tests/benchmarks/CMakeLists.txt | 36 ++++---- tests/benchmarks/README.md | 130 ++++++++++++++++++++++++++++ tests/benchmarks/src/benchmarks.cpp | 3 +- 5 files changed, 155 insertions(+), 18 deletions(-) create mode 100644 tests/benchmarks/README.md diff --git a/cmake/ci.cmake b/cmake/ci.cmake index 752bdc6f8..573a8e0e0 100644 --- a/cmake/ci.cmake +++ b/cmake/ci.cmake @@ -619,7 +619,7 @@ add_custom_target(ci_single_binaries add_custom_target(ci_benchmarks COMMAND ${CMAKE_COMMAND} -DCMAKE_BUILD_TYPE=Release -GNinja - -S${PROJECT_SOURCE_DIR}/benchmarks -B${PROJECT_BINARY_DIR}/build_benchmarks + -S${PROJECT_SOURCE_DIR}/tests/benchmarks -B${PROJECT_BINARY_DIR}/build_benchmarks COMMAND ${CMAKE_COMMAND} --build ${PROJECT_BINARY_DIR}/build_benchmarks --target json_benchmarks COMMAND cd ${PROJECT_BINARY_DIR}/build_benchmarks && ./json_benchmarks COMMENT "Run benchmarks" diff --git a/cmake/download_test_data.cmake b/cmake/download_test_data.cmake index b7211ac54..3b6f9a395 100644 --- a/cmake/download_test_data.cmake +++ b/cmake/download_test_data.cmake @@ -77,7 +77,7 @@ if(CMAKE_CROSSCOMPILING) endif() if(NOT DEFINED LIBCPP_VERSION_OUTPUT_CACHED) try_run(RUN_RESULT_VAR COMPILE_RESULT_VAR - "${CMAKE_BINARY_DIR}" SOURCES "${CMAKE_SOURCE_DIR}/cmake/detect_libcpp_version.cpp" + "${CMAKE_BINARY_DIR}" SOURCES "${CMAKE_CURRENT_LIST_DIR}/detect_libcpp_version.cpp" RUN_OUTPUT_VARIABLE LIBCPP_VERSION_OUTPUT COMPILE_OUTPUT_VARIABLE LIBCPP_VERSION_COMPILE_OUTPUT ) diff --git a/tests/benchmarks/CMakeLists.txt b/tests/benchmarks/CMakeLists.txt index 4aa095e39..4d7265145 100644 --- a/tests/benchmarks/CMakeLists.txt +++ b/tests/benchmarks/CMakeLists.txt @@ -1,4 +1,4 @@ -cmake_minimum_required(VERSION 3.11...3.14) +cmake_minimum_required(VERSION 3.14) project(JSON_Benchmarks LANGUAGES CXX) # set compiler flags @@ -6,29 +6,35 @@ if((CMAKE_CXX_COMPILER_ID MATCHES GNU) OR (CMAKE_CXX_COMPILER_ID MATCHES Clang)) set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -flto -DNDEBUG -O3") endif() -# configure Google Benchmarks +# configure Google Benchmark; a fixed release, so that results stay comparable +set(JSON_GOOGLE_BENCHMARK_VERSION 1.9.5) include(FetchContent) -FetchContent_Declare( - benchmark - GIT_REPOSITORY https://github.com/google/benchmark.git - GIT_TAG origin/main - GIT_SHALLOW TRUE -) -FetchContent_GetProperties(benchmark) -if(NOT benchmark_POPULATED) - FetchContent_Populate(benchmark) - set(BENCHMARK_ENABLE_TESTING OFF CACHE INTERNAL "" FORCE) - add_subdirectory(${benchmark_SOURCE_DIR} ${benchmark_BINARY_DIR}) -endif() +# only the library is needed; -Werror would break the pinned release as soon as +# a newer compiler adds a warning +set(BENCHMARK_ENABLE_TESTING OFF CACHE BOOL "" FORCE) +set(BENCHMARK_ENABLE_INSTALL OFF CACHE BOOL "" FORCE) +set(BENCHMARK_ENABLE_WERROR OFF CACHE BOOL "" FORCE) + +FetchContent_Declare(benchmark + URL https://github.com/google/benchmark/archive/refs/tags/v${JSON_GOOGLE_BENCHMARK_VERSION}.tar.gz + URL_HASH SHA256=9631341c82bac4a288bef951f8b26b41f69021794184ece969f8473977eaa340 + DOWNLOAD_EXTRACT_TIMESTAMP TRUE +) +FetchContent_MakeAvailable(benchmark) # download test data set(CMAKE_MODULE_PATH ${CMAKE_CURRENT_SOURCE_DIR}/../../cmake ${CMAKE_MODULE_PATH}) include(download_test_data) +# the header to benchmark; point this at a directory holding another version's +# nlohmann/json.hpp to compare versions (see README.md) +set(JSON_BENCHMARK_INCLUDE_DIR "${CMAKE_CURRENT_SOURCE_DIR}/../../single_include" CACHE PATH + "directory containing the nlohmann/json.hpp to benchmark") + # benchmark binary add_executable(json_benchmarks src/benchmarks.cpp) target_compile_features(json_benchmarks PRIVATE cxx_std_11) target_link_libraries(json_benchmarks benchmark ${CMAKE_THREAD_LIBS_INIT}) add_dependencies(json_benchmarks download_test_data) -target_include_directories(json_benchmarks PRIVATE ${CMAKE_SOURCE_DIR}/../../single_include ${CMAKE_BINARY_DIR}/include) +target_include_directories(json_benchmarks PRIVATE ${JSON_BENCHMARK_INCLUDE_DIR} ${CMAKE_BINARY_DIR}/include) diff --git a/tests/benchmarks/README.md b/tests/benchmarks/README.md new file mode 100644 index 000000000..aaf5abe46 --- /dev/null +++ b/tests/benchmarks/README.md @@ -0,0 +1,130 @@ +# Benchmarks + +Micro-benchmarks for parsing, serialization and the binary formats, written with +[Google Benchmark](https://github.com/google/benchmark). They are not run by CI; see +[When to run them](#when-to-run-them). + +## What is measured + +| benchmark | what it does | +|---|---| +| `ParseFile`, `ParseString` | parse JSON from a file stream or a string | +| `ParseIndented` | parse the large files re-indented by 4 spaces, for the lexer's whitespace handling | +| `Dump` | serialize, compact (`-`) and indented (`4`) | +| `ToCbor`, `BinaryToCbor` | write CBOR; `BinaryToCbor` writes binary values of growing size | +| `FromMsgpack` | read MessagePack; unchanged over the years, so its numbers stay comparable across releases | +| `FromBinaryBuffer`, `FromBinaryFile` | read CBOR, MessagePack, UBJSON, BJData and BSON from a buffer or a `FILE*` | +| `FromBinaryShape` | read deeply nested, container-heavy and scalar-heavy documents in every binary format | +| `FromCborChunkedString` | read CBOR strings split into indefinite-length chunks | + +The input files are those of [nativejson-benchmark](https://github.com/miloyip/nativejson-benchmark) (`canada`, +`citm_catalog`, `twitter`), a large `jeopardy` file, and number-heavy files (`floats`, `signed_ints`, ...). +`bytes_per_second` counts the bytes read or written: the JSON text when parsing, the output when serializing. + +## Requirements + +- CMake 3.14 or later, a C++11 compiler, and Ninja for the `make` target. +- Network access on the first configure: CMake downloads Google Benchmark and the + [test data](https://github.com/nlohmann/json_test_data) into the build directory. To reuse a download of the test + data, pass `-DJSON_TestDataDirectory=/test_files`. +- Google Benchmark is pinned to a release (1.9.5), so that results from different days stay comparable. To update it, + change `JSON_GOOGLE_BENCHMARK_VERSION` and the archive's `URL_HASH` in `CMakeLists.txt` together. +- The benchmarks include `single_include/nlohmann/json.hpp`, so run `make amalgamate` after changing anything in + `include/`. + +GCC and Clang builds use `-O3 -flto -DNDEBUG`. + +## Running them + +From the repository root, this builds everything from scratch in `cmake-build-benchmarks` and runs all benchmarks: + +```sh +make run_benchmarks +``` + +To build once and run selectively: + +```sh +cmake -S tests/benchmarks -B build-benchmarks -G Ninja -DCMAKE_BUILD_TYPE=Release +cmake --build build-benchmarks +build-benchmarks/json_benchmarks --benchmark_filter='ParseString|Dump' +``` + +Useful options of `json_benchmarks`: + +| option | effect | +|---|---| +| `--benchmark_list_tests` | list the benchmarks instead of running them | +| `--benchmark_filter=` | run only the benchmarks whose names match | +| `--benchmark_repetitions=` | run every benchmark `n` times and add mean, median, standard deviation and coefficient of variation | +| `--benchmark_enable_random_interleaving=true` | run the repetitions in random order, which spreads out drifts such as thermal throttling | +| `--benchmark_min_time=s` | run each benchmark at least this long (e.g. `2s`) | +| `--benchmark_out= --benchmark_out_format=json` | also write the results to a file, e.g. for `compare.py` | + +## Reading the output + +Each line shows the wall-clock `Time` and the `CPU` time per iteration, the number of `Iterations` Google Benchmark +chose, and the throughput in `bytes_per_second`. With repetitions, the lines ending in `_median` are the ones to +compare. A `_cv` (coefficient of variation) above a few percent means the machine was too noisy for small +differences to mean anything. + +## Comparing two versions + +To see what a change or a release did, build the same benchmarks twice: once against the header of the version to +compare with, and once against the current one. `JSON_BENCHMARK_INCLUDE_DIR` names the directory holding the +`nlohmann/json.hpp` to benchmark. For example, to compare the current checkout with 3.12.0: + +```sh +# the header of the version to compare with +mkdir -p build-baseline-header/nlohmann +git show v3.12.0:single_include/nlohmann/json.hpp > build-baseline-header/nlohmann/json.hpp + +# the same benchmarks, built against either header +cmake -S tests/benchmarks -B build-baseline -G Ninja -DCMAKE_BUILD_TYPE=Release \ + -DJSON_BENCHMARK_INCLUDE_DIR="$PWD/build-baseline-header" +cmake -S tests/benchmarks -B build-current -G Ninja -DCMAKE_BUILD_TYPE=Release +cmake --build build-baseline +cmake --build build-current + +# run both, back to back +build-baseline/json_benchmarks --benchmark_repetitions=10 --benchmark_enable_random_interleaving=true \ + --benchmark_out=build-baseline/results.json --benchmark_out_format=json +build-current/json_benchmarks --benchmark_repetitions=10 --benchmark_enable_random_interleaving=true \ + --benchmark_out=build-current/results.json --benchmark_out_format=json +``` + +Google Benchmark ships a tool to compare the two result files. It needs NumPy and SciPy: + +```sh +python3 -m venv build-venv +build-venv/bin/pip install numpy scipy +build-venv/bin/python build-current/_deps/benchmark-src/tools/compare.py -a benchmarks build-baseline/results.json build-current/results.json +``` + +The tool's own `tools/requirements.txt` pins NumPy and SciPy versions that need Python 3.11 or later; with an older +Python, unpinned versions work as well. In its output: + +- the `Time` and `CPU` columns are relative changes: `-0.35` means 35% faster, `+0.10` means 10% slower; +- `_pvalue` lines report a Mann-Whitney U test of whether the two versions differ. It needs at least 9 + repetitions, and a p-value below 0.05 means the difference is unlikely to be noise; +- `OVERALL_GEOMEAN` summarizes all benchmarks; +- `-a` shows only the aggregates, not every repetition. + +The header you compare with must support everything the benchmarks use. The current benchmarks build against 3.12.0. +Only benchmarks present in both result files are compared, so for older releases, either filter the benchmarks or +build that release's own `tests/benchmarks` against its own header. + +## Getting stable numbers + +- Build and run both versions on the same machine, one right after the other. +- Keep the machine otherwise idle: no builds, no browser, and a laptop plugged in. +- On Linux, set the CPU frequency governor to `performance`, e.g. `sudo cpupower frequency-set --governor performance`. + Google Benchmark prints a warning when frequency scaling is enabled. Pinning the process to a core + (`taskset -c 2 ...`) helps as well. +- Use 10 or more repetitions with random interleaving, compare medians, and treat changes within the `_cv` as noise. + +## When to run them + +They are a manual step, not part of CI: shared CI runners vary more between runs than most of the effects measured. +Run the comparison above before a release, comparing the previous release tag with `develop`, and for pull requests +that claim to change performance. diff --git a/tests/benchmarks/src/benchmarks.cpp b/tests/benchmarks/src/benchmarks.cpp index f949f3f12..613ca1baa 100644 --- a/tests/benchmarks/src/benchmarks.cpp +++ b/tests/benchmarks/src/benchmarks.cpp @@ -131,7 +131,8 @@ static void Dump(benchmark::State& state, const char* filename, int indent) while (state.KeepRunning()) { - j.dump(indent); + std::string output = j.dump(indent); + benchmark::DoNotOptimize(output); } state.SetBytesProcessed(state.iterations() * j.dump(indent).size()); From 632a5812a8ee922f198a1eef68966b344cdd985b Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 25 Sep 2026 20:44:30 +0200 Subject: [PATCH 49/64] Support zero-member types in NLOHMANN_DEFINE_TYPE_* macros (#4041) (#5272) * Support zero-member types in NLOHMANN_DEFINE_TYPE_* macros (#4041) NLOHMANN_DEFINE_TYPE_INTRUSIVE(Type) and its 11 sibling macros produced broken code for types with no members to serialize. Invoking a variadic macro so __VA_ARGS__ is empty is only standard-conforming since C++20, so a plain __VA_OPT__ fix (as tried in #5142) breaks every pre-C++20 build under -pedantic. Instead, make all 12 macros purely variadic and dispatch on argument count using a sentinel-padded extension of the existing NLOHMANN_JSON_GET_MACRO idiom, giving full C++11-C++26 support with no feature-test gate. Verified against real GCC 16 and Clang at -std=c++11/14/17/20 with -pedantic -Werror -Wvariadic-macros: zero regressions in the existing unit-udt_macro.cpp suite plus 12 new zero-member test cases. Signed-off-by: Niels Lohmann * Fix CI failures in zero-member NLOHMANN_DEFINE_TYPE_* macros Three issues surfaced on PR #5272's real CI that weren't caught by local testing against a narrower flag set: - GCC -Werror=noexcept: the four truly-empty from_json bodies (plain INTRUSIVE/NON_INTRUSIVE, with and without _WITH_DEFAULT) provably never throw but weren't declared noexcept; mark them noexcept explicitly. to_json and the derived-type from_json overloads are left alone since they genuinely can throw (object assignment / delegating to the base class's from_json). - clang-tidy bugprone-macro-parentheses: false positive on the same 8 zero-member bodies (Type/BaseType used purely as declarator types); suppressed with NOLINTNEXTLINE comments in the same style already used elsewhere in this file (see NLOHMANN_JSON_SERIALIZE_ENUM). - MSVC's traditional preprocessor doesn't fully expand NLOHMANN_JSON_CAT(prefix, NLOHMANN_JSON_TYPE_TAG(...))(...) in one pass, which broke a pre-existing one-member usage in unit-regression2.cpp with syntax errors. Wrap all 12 public dispatcher macros in an extra outer NLOHMANN_JSON_EXPAND(...), matching the pattern NLOHMANN_JSON_PASTE already uses for the same MSVC quirk. Re-verified against real GCC 16 and Clang at -std=c++11/14/17/20 with -pedantic -Werror -Wvariadic-macros -Wnoexcept, including the exact files that failed in CI (unit-udt_macro.cpp, unit-regression2.cpp), against both the modular headers and the re-amalgamated single header. Signed-off-by: Niels Lohmann * Fix clang-tidy misc-const-correctness in unit-udt_macro.cpp The four zero-member ONLY_SERIALIZE test objects are only ever read (via to_json), never mutated, so mark them const per clang-tidy. Signed-off-by: Niels Lohmann * Fix derived-type macro dispatch capping members at 62 instead of 63 NLOHMANN_JSON_GET_MACRO resolves 64 positional arguments, with NAME at position 65. NLOHMANN_JSON_TYPE_TAG dispatches on Type plus the member list, so it resolves correctly up to the 63 members NLOHMANN_JSON_PASTE supports. NLOHMANN_JSON_DERIVED_TYPE_TAG dispatched on the two-token Type,BaseType prefix plus the member list, running out one slot early: at 63 members, position 65 landed on the last member name instead of a sentinel and NLOHMANN_JSON_CAT built an undefined identifier such as NLOHMANN_JSON_DEFINE_DERIVED_TYPE_INTRUSIVE_m63, with the compiler reporting "unknown type name 'm1'" once per member and nothing pointing at an argument-count limit. That silently reduced all six NLOHMANN_DEFINE_DERIVED_TYPE_* macros from 63 members to 62, contradicting the "up to 63 members" contract in docs/mkdocs/docs/api/macros/nlohmann_define_derived_type.md. Drop the leading Type and defer to NLOHMANN_JSON_TYPE_TAG so the tag is computed from BaseType plus the member list, which fits the available slots. The zero-own-member derived bodies are therefore selected by tag 1 rather than 2, and the sentinel table for the derived tag is no longer needed. Add a regression test at the documented maximum for both the plain and the derived macros; it fails to compile against the previous dispatch. Signed-off-by: Niels Lohmann * Name the zero-member macro bodies by intent, not argument count The dispatch tag was the literal token 1 or N, pasted onto a macro prefix to select the zero-member or member-carrying body. For the derived-type macros that reads wrong: their tag is computed after dropping the leading Type, so the zero-member body was named _1 while taking two parameters (Type, BaseType). Emit EMPTY and MEMBERS instead. The mechanism is unchanged -- the tag is still a token pasted onto the prefix by NLOHMANN_JSON_CAT -- but the body names now say what they are rather than encoding an argument count that only lines up for half of the macros. Collapse the four duplicated zero-member bodies while here: with no members there is nothing to default, so each _WITH_DEFAULT_EMPTY body was a byte-for-byte copy of its plain counterpart. They are now one-line aliases, leaving a single definition of what an empty object serializes to per intrusive/non-intrusive and base/derived combination. No functional change: for both zero-member and member-carrying types the preprocessed to_json/from_json output is token-for-token identical, and the arity limits are unchanged (63 members, base and derived). Signed-off-by: Niels Lohmann * Document zero-member support in the macro API reference docs/mkdocs/docs/features/arbitrary_types.md already gained a note, but the three api/macros pages are where the parameter contract is actually specified and they still described member as a non-empty list. State that the list may be empty on each page, and add a note showing what the zero-member case generates: an empty JSON object for the plain macros, and base-type-only serialization for the derived ones. Both notes record that the WITH_NAMES variants do not support this. Signed-off-by: Niels Lohmann * Keep user macros named EMPTY or MEMBERS out of the member-count dispatch The dispatch produced the bare token EMPTY or MEMBERS and pasted it onto the macro prefix afterwards. In between, the token was rescanned, so a user macro with either name replaced it: with `#define MEMBERS x` in scope, even NLOHMANN_DEFINE_TYPE_INTRUSIVE(A, member) -- which compiled before -- expanded to garbage, and `#define EMPTY` broke the zero-member form. Paste the suffix onto the prefix directly in the GET_MACRO slot table instead. Operands of ## are not macro-expanded, so the selected body name is formed before any user macro can interfere. NLOHMANN_JSON_TYPE_TAG and NLOHMANN_JSON_DERIVED_TYPE_TAG become NLOHMANN_JSON_TYPE_BODY and NLOHMANN_JSON_DERIVED_TYPE_BODY, taking the prefix as their first argument; NLOHMANN_JSON_CAT is no longer needed. The body macro names are unchanged, and so is the generated code. Add a regression test that defines EMPTY and MEMBERS around plain and derived types, with and without members. Signed-off-by: Niels Lohmann * Test for EMPTY and MEMBERS so -Wunused-macros accepts them Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- .../macros/nlohmann_define_derived_type.md | 17 +- .../macros/nlohmann_define_type_intrusive.md | 17 +- .../nlohmann_define_type_non_intrusive.md | 15 +- docs/mkdocs/docs/features/arbitrary_types.md | 18 + include/nlohmann/detail/macro_scope.hpp | 131 ++++++- single_include/nlohmann/json.hpp | 131 ++++++- tests/src/unit-udt_macro.cpp | 367 ++++++++++++++++++ 7 files changed, 669 insertions(+), 27 deletions(-) diff --git a/docs/mkdocs/docs/api/macros/nlohmann_define_derived_type.md b/docs/mkdocs/docs/api/macros/nlohmann_define_derived_type.md index 433d5b9a0..8c549d137 100644 --- a/docs/mkdocs/docs/api/macros/nlohmann_define_derived_type.md +++ b/docs/mkdocs/docs/api/macros/nlohmann_define_derived_type.md @@ -57,7 +57,8 @@ Summary: : name of the base type (class, struct) `type` is derived from `member` (in) -: name of the member variable to serialize/deserialize; up to 63 members can be given as a comma-separated list +: name of the member variable to serialize/deserialize; up to 63 members can be given as a comma-separated + list, which may also be empty ## Default definition @@ -127,6 +128,20 @@ void to_json(BasicJsonType& j, const B& b) { - Macros 4, 5, and 6 have the same prerequisites of [NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE](nlohmann_define_type_non_intrusive.md). - Serialization/deserialization of base types must be defined. +!!! info "Derived types without own members" + + The member list may be empty. The macro then generates a `to_json`/`from_json` pair that only delegates to + the base type, so `type` serializes exactly like `base_type`: + + ```cpp + struct derived : base + { + NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE(derived, base) + }; + ``` + + The `WITH_NAMES` variants do not support this. + !!! warning "Implementation limits" See Implementation limits for [NLOHMANN_DEFINE_TYPE_INTRUSIVE](nlohmann_define_type_intrusive.md) and diff --git a/docs/mkdocs/docs/api/macros/nlohmann_define_type_intrusive.md b/docs/mkdocs/docs/api/macros/nlohmann_define_type_intrusive.md index f0c32c557..deef3fec9 100644 --- a/docs/mkdocs/docs/api/macros/nlohmann_define_type_intrusive.md +++ b/docs/mkdocs/docs/api/macros/nlohmann_define_type_intrusive.md @@ -33,7 +33,8 @@ Summary: : name of the type (class, struct) to serialize/deserialize `member` (in) -: name of the member variable to serialize/deserialize; up to 63 members can be given as a comma-separated list +: name of the member variable to serialize/deserialize; up to 63 members can be given as a comma-separated + list, which may also be empty ## Default definition @@ -58,6 +59,20 @@ See the examples below for the concrete generated code. [GetNonDefNonCopy]: ../../features/arbitrary_types.md#how-can-i-use-get-for-non-default-constructiblenon-copyable-types +!!! info "Types without members" + + The member list may be empty. The macro then generates a `to_json` that produces an empty JSON object + `#!json {}`, and a `from_json` that reads no members: + + ```cpp + struct marker + { + NLOHMANN_DEFINE_TYPE_INTRUSIVE(marker) + }; + ``` + + The `WITH_NAMES` variants do not support this. + !!! warning "Implementation limits" - The current implementation is limited to at most 63 member variables. If you want to serialize/deserialize types diff --git a/docs/mkdocs/docs/api/macros/nlohmann_define_type_non_intrusive.md b/docs/mkdocs/docs/api/macros/nlohmann_define_type_non_intrusive.md index cc8304750..6a29c9e66 100644 --- a/docs/mkdocs/docs/api/macros/nlohmann_define_type_non_intrusive.md +++ b/docs/mkdocs/docs/api/macros/nlohmann_define_type_non_intrusive.md @@ -33,7 +33,8 @@ Summary: : name of the type (class, struct) to serialize/deserialize `member` (in) -: name of the (public) member variable to serialize/deserialize; up to 63 members can be given as a comma-separated list +: name of the (public) member variable to serialize/deserialize; up to 63 members can be given as a + comma-separated list, which may also be empty ## Default definition @@ -59,6 +60,18 @@ See the examples below for the concrete generated code. [GetNonDefNonCopy]: ../../features/arbitrary_types.md#how-can-i-use-get-for-non-default-constructiblenon-copyable-types +!!! info "Types without members" + + The member list may be empty. The macro then generates a `to_json` that produces an empty JSON object + `#!json {}`, and a `from_json` that reads no members: + + ```cpp + struct marker {}; + NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE(marker) + ``` + + The `WITH_NAMES` variants do not support this. + !!! warning "Implementation limits" - The current implementation is limited to at most 63 member variables. If you want to serialize/deserialize types diff --git a/docs/mkdocs/docs/features/arbitrary_types.md b/docs/mkdocs/docs/features/arbitrary_types.md index b472861b7..7ea01ee55 100644 --- a/docs/mkdocs/docs/features/arbitrary_types.md +++ b/docs/mkdocs/docs/features/arbitrary_types.md @@ -215,6 +215,24 @@ For _derived_ classes and structs, use the following macros nlohmann::ordered_json j = p; // keys appear in declaration order: name, address, age ``` +!!! note "Zero-member types" + + All 12 `NLOHMANN_DEFINE_TYPE_*`/`NLOHMANN_DEFINE_DERIVED_TYPE_*` macros (excluding the `WITH_NAMES` variants) + also accept types with no member variables to serialize, producing/accepting an empty JSON object `{}` + (or, for the derived-type macros, just the base class's own JSON representation): + + ```cpp + namespace ns { + struct marker { + bool operator==(const marker&) const { return true; } + NLOHMANN_DEFINE_TYPE_INTRUSIVE(marker) + }; + } + + ns::marker m{}; + nlohmann::json j = m; // {} + ``` + !!! note "No macro for non-default-constructible types" There is currently no `NLOHMANN_DEFINE_TYPE_*`-style macro for types that are not diff --git a/include/nlohmann/detail/macro_scope.hpp b/include/nlohmann/detail/macro_scope.hpp index 96fa165f5..9094aa6e4 100644 --- a/include/nlohmann/detail/macro_scope.hpp +++ b/include/nlohmann/detail/macro_scope.hpp @@ -596,16 +596,62 @@ void templated_json_throw(ExceptionType exception) #define NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(...) template::value, int> = 0> __VA_ARGS__ +// Helpers used to dispatch the NLOHMANN_DEFINE_TYPE_*/NLOHMANN_DEFINE_DERIVED_TYPE_* +// macros below between a zero-member and a one-or-more-member implementation +// (issue #4041, e.g. NLOHMANN_DEFINE_TYPE_INTRUSIVE(Type) with no further +// arguments). NLOHMANN_JSON_TYPE_BODY(Prefix, ...) expands to the macro name +// Prefix##EMPTY when __VA_ARGS__ is a single argument (Type alone) and to +// Prefix##MEMBERS for two or more (Type, member...). It reuses the existing +// 64-slot NLOHMANN_JSON_GET_MACRO dispatch with one extra trailing sentinel +// token appended so its own trailing "..." is never left completely empty at +// the lowest supported argument count -- invoking a variadic macro so that +// "..." matches nothing is only granted unconditionally by the standard since +// C++20, and pre-C++20 compilers may reject it under -pedantic regardless of +// what the macro body does. +// +// The EMPTY/MEMBERS suffixes are pasted onto Prefix right in the slot table: +// operands of ## are not macro-expanded, so the dispatch keeps working even if +// user code defines macros named EMPTY or MEMBERS. Producing the bare suffix +// first and pasting it later would let such a macro replace it. +// +// NLOHMANN_JSON_DERIVED_TYPE_BODY(Prefix, ...) answers the same question for +// the derived-type macros, whose fixed prefix is Type,BaseType. It drops the +// leading Type and defers to NLOHMANN_JSON_TYPE_BODY rather than shifting the +// slot table by one: NLOHMANN_JSON_GET_MACRO only resolves 64 positional +// arguments, so dispatching on Type,BaseType,member... directly would run out +// one slot early and cap the derived-type macros at 62 members instead of the +// 63 that NLOHMANN_JSON_PASTE supports. +#define NLOHMANN_JSON_TYPE_BODY(Prefix, ...) NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_GET_MACRO(__VA_ARGS__, \ + Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, \ + Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, \ + Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, \ + Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, \ + Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, \ + Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, \ + Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, \ + Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## EMPTY, \ + NLOHMANN_JSON_TYPE_BODY_SENTINEL)) + +#define NLOHMANN_JSON_DERIVED_TYPE_BODY_(Prefix, Type, ...) NLOHMANN_JSON_TYPE_BODY(Prefix, __VA_ARGS__) +#define NLOHMANN_JSON_DERIVED_TYPE_BODY(Prefix, ...) NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_DERIVED_TYPE_BODY_(Prefix, __VA_ARGS__)) + /*! @brief macro @def NLOHMANN_DEFINE_TYPE_INTRUSIVE @since version 3.9.0 @sa https://json.nlohmann.me/api/macros/nlohmann_define_type_intrusive/ */ -#define NLOHMANN_DEFINE_TYPE_INTRUSIVE(Type, ...) \ +#define NLOHMANN_JSON_DEFINE_TYPE_INTRUSIVE_MEMBERS(Type, ...) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(friend void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_PASTE(NLOHMANN_JSON_TO, __VA_ARGS__)) }) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(friend void from_json(const BasicJsonType& nlohmann_json_j, Type& nlohmann_json_t) { NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_PASTE(NLOHMANN_JSON_FROM, __VA_ARGS__)) }) +#define NLOHMANN_JSON_DEFINE_TYPE_INTRUSIVE_EMPTY(Type) \ + NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(friend void to_json(BasicJsonType& nlohmann_json_j, const Type&) { nlohmann_json_j = BasicJsonType::object(); }) \ + /* NOLINTNEXTLINE(bugprone-macro-parentheses) Type is used as a declarator type, not in an expression */ \ + NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(friend void from_json(const BasicJsonType&, Type&) noexcept { }) + +#define NLOHMANN_DEFINE_TYPE_INTRUSIVE(...) NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_TYPE_BODY(NLOHMANN_JSON_DEFINE_TYPE_INTRUSIVE_, __VA_ARGS__)(__VA_ARGS__)) + #define NLOHMANN_DEFINE_TYPE_INTRUSIVE_WITH_NAMES(Type, ...) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(friend void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_DOUBLE_PASTE(NLOHMANN_JSON_TO_WITH_NAME, __VA_ARGS__)) }) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(friend void from_json(const BasicJsonType& nlohmann_json_j, Type& nlohmann_json_t) { NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_DOUBLE_PASTE(NLOHMANN_JSON_FROM_WITH_NAME, __VA_ARGS__)) }) @@ -616,10 +662,15 @@ void templated_json_throw(ExceptionType exception) @since version 3.11.0 @sa https://json.nlohmann.me/api/macros/nlohmann_define_type_intrusive/ */ -#define NLOHMANN_DEFINE_TYPE_INTRUSIVE_WITH_DEFAULT(Type, ...) \ +#define NLOHMANN_JSON_DEFINE_TYPE_INTRUSIVE_WITH_DEFAULT_MEMBERS(Type, ...) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(friend void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_PASTE(NLOHMANN_JSON_TO, __VA_ARGS__)) }) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(friend void from_json(const BasicJsonType& nlohmann_json_j, Type& nlohmann_json_t) { const Type nlohmann_json_default_obj{}; NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_PASTE(NLOHMANN_JSON_FROM_WITH_DEFAULT, __VA_ARGS__)) }) +// identical to NLOHMANN_JSON_DEFINE_TYPE_INTRUSIVE_EMPTY: with no members there is nothing to default +#define NLOHMANN_JSON_DEFINE_TYPE_INTRUSIVE_WITH_DEFAULT_EMPTY(Type) NLOHMANN_JSON_DEFINE_TYPE_INTRUSIVE_EMPTY(Type) + +#define NLOHMANN_DEFINE_TYPE_INTRUSIVE_WITH_DEFAULT(...) NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_TYPE_BODY(NLOHMANN_JSON_DEFINE_TYPE_INTRUSIVE_WITH_DEFAULT_, __VA_ARGS__)(__VA_ARGS__)) + #define NLOHMANN_DEFINE_TYPE_INTRUSIVE_WITH_DEFAULT_WITH_NAMES(Type, ...) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(friend void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_DOUBLE_PASTE(NLOHMANN_JSON_TO_WITH_NAME, __VA_ARGS__)) }) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(friend void from_json(const BasicJsonType& nlohmann_json_j, Type& nlohmann_json_t) { const Type nlohmann_json_default_obj{}; NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_DOUBLE_PASTE(NLOHMANN_JSON_FROM_WITH_DEFAULT_WITH_NAME, __VA_ARGS__)) }) @@ -630,9 +681,14 @@ void templated_json_throw(ExceptionType exception) @since version 3.11.3 @sa https://json.nlohmann.me/api/macros/nlohmann_define_type_intrusive/ */ -#define NLOHMANN_DEFINE_TYPE_INTRUSIVE_ONLY_SERIALIZE(Type, ...) \ +#define NLOHMANN_JSON_DEFINE_TYPE_INTRUSIVE_ONLY_SERIALIZE_MEMBERS(Type, ...) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(friend void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_PASTE(NLOHMANN_JSON_TO, __VA_ARGS__)) }) +#define NLOHMANN_JSON_DEFINE_TYPE_INTRUSIVE_ONLY_SERIALIZE_EMPTY(Type) \ + NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(friend void to_json(BasicJsonType& nlohmann_json_j, const Type&) { nlohmann_json_j = BasicJsonType::object(); }) + +#define NLOHMANN_DEFINE_TYPE_INTRUSIVE_ONLY_SERIALIZE(...) NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_TYPE_BODY(NLOHMANN_JSON_DEFINE_TYPE_INTRUSIVE_ONLY_SERIALIZE_, __VA_ARGS__)(__VA_ARGS__)) + #define NLOHMANN_DEFINE_TYPE_INTRUSIVE_ONLY_SERIALIZE_WITH_NAMES(Type, ...) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(friend void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_DOUBLE_PASTE(NLOHMANN_JSON_TO_WITH_NAME, __VA_ARGS__)) }) @@ -642,10 +698,17 @@ void templated_json_throw(ExceptionType exception) @since version 3.9.0 @sa https://json.nlohmann.me/api/macros/nlohmann_define_type_non_intrusive/ */ -#define NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE(Type, ...) \ +#define NLOHMANN_JSON_DEFINE_TYPE_NON_INTRUSIVE_MEMBERS(Type, ...) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_PASTE(NLOHMANN_JSON_TO, __VA_ARGS__)) }) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(void from_json(const BasicJsonType& nlohmann_json_j, Type& nlohmann_json_t) { NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_PASTE(NLOHMANN_JSON_FROM, __VA_ARGS__)) }) +#define NLOHMANN_JSON_DEFINE_TYPE_NON_INTRUSIVE_EMPTY(Type) \ + NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(void to_json(BasicJsonType& nlohmann_json_j, const Type&) { nlohmann_json_j = BasicJsonType::object(); }) \ + /* NOLINTNEXTLINE(bugprone-macro-parentheses) Type is used as a declarator type, not in an expression */ \ + NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(void from_json(const BasicJsonType&, Type&) noexcept { }) + +#define NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE(...) NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_TYPE_BODY(NLOHMANN_JSON_DEFINE_TYPE_NON_INTRUSIVE_, __VA_ARGS__)(__VA_ARGS__)) + #define NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE_WITH_NAMES(Type, ...) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_DOUBLE_PASTE(NLOHMANN_JSON_TO_WITH_NAME, __VA_ARGS__)) }) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(void from_json(const BasicJsonType& nlohmann_json_j, Type& nlohmann_json_t) { NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_DOUBLE_PASTE(NLOHMANN_JSON_FROM_WITH_NAME, __VA_ARGS__)) }) @@ -656,10 +719,15 @@ void templated_json_throw(ExceptionType exception) @since version 3.11.0 @sa https://json.nlohmann.me/api/macros/nlohmann_define_type_non_intrusive/ */ -#define NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE_WITH_DEFAULT(Type, ...) \ +#define NLOHMANN_JSON_DEFINE_TYPE_NON_INTRUSIVE_WITH_DEFAULT_MEMBERS(Type, ...) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_PASTE(NLOHMANN_JSON_TO, __VA_ARGS__)) }) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(void from_json(const BasicJsonType& nlohmann_json_j, Type& nlohmann_json_t) { const Type nlohmann_json_default_obj{}; NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_PASTE(NLOHMANN_JSON_FROM_WITH_DEFAULT, __VA_ARGS__)) }) +// identical to NLOHMANN_JSON_DEFINE_TYPE_NON_INTRUSIVE_EMPTY: with no members there is nothing to default +#define NLOHMANN_JSON_DEFINE_TYPE_NON_INTRUSIVE_WITH_DEFAULT_EMPTY(Type) NLOHMANN_JSON_DEFINE_TYPE_NON_INTRUSIVE_EMPTY(Type) + +#define NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE_WITH_DEFAULT(...) NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_TYPE_BODY(NLOHMANN_JSON_DEFINE_TYPE_NON_INTRUSIVE_WITH_DEFAULT_, __VA_ARGS__)(__VA_ARGS__)) + #define NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE_WITH_DEFAULT_WITH_NAMES(Type, ...) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_DOUBLE_PASTE(NLOHMANN_JSON_TO_WITH_NAME, __VA_ARGS__)) }) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(void from_json(const BasicJsonType& nlohmann_json_j, Type& nlohmann_json_t) { const Type nlohmann_json_default_obj{}; NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_DOUBLE_PASTE(NLOHMANN_JSON_FROM_WITH_DEFAULT_WITH_NAME, __VA_ARGS__)) }) @@ -670,9 +738,14 @@ void templated_json_throw(ExceptionType exception) @since version 3.11.3 @sa https://json.nlohmann.me/api/macros/nlohmann_define_type_non_intrusive/ */ -#define NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE_ONLY_SERIALIZE(Type, ...) \ +#define NLOHMANN_JSON_DEFINE_TYPE_NON_INTRUSIVE_ONLY_SERIALIZE_MEMBERS(Type, ...) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_PASTE(NLOHMANN_JSON_TO, __VA_ARGS__)) }) +#define NLOHMANN_JSON_DEFINE_TYPE_NON_INTRUSIVE_ONLY_SERIALIZE_EMPTY(Type) \ + NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(void to_json(BasicJsonType& nlohmann_json_j, const Type&) { nlohmann_json_j = BasicJsonType::object(); }) + +#define NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE_ONLY_SERIALIZE(...) NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_TYPE_BODY(NLOHMANN_JSON_DEFINE_TYPE_NON_INTRUSIVE_ONLY_SERIALIZE_, __VA_ARGS__)(__VA_ARGS__)) + #define NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE_ONLY_SERIALIZE_WITH_NAMES(Type, ...) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_DOUBLE_PASTE(NLOHMANN_JSON_TO_WITH_NAME, __VA_ARGS__)) }) @@ -682,10 +755,17 @@ void templated_json_throw(ExceptionType exception) @since version 3.12.0 @sa https://json.nlohmann.me/api/macros/nlohmann_define_derived_type/ */ -#define NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE(Type, BaseType, ...) \ +#define NLOHMANN_JSON_DEFINE_DERIVED_TYPE_INTRUSIVE_MEMBERS(Type, BaseType, ...) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(friend void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { nlohmann::to_json(nlohmann_json_j, static_cast(nlohmann_json_t)); NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_PASTE(NLOHMANN_JSON_TO, __VA_ARGS__)) }) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(friend void from_json(const BasicJsonType& nlohmann_json_j, Type& nlohmann_json_t) { nlohmann::from_json(nlohmann_json_j, static_cast(nlohmann_json_t)); NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_PASTE(NLOHMANN_JSON_FROM, __VA_ARGS__)) }) +#define NLOHMANN_JSON_DEFINE_DERIVED_TYPE_INTRUSIVE_EMPTY(Type, BaseType) \ + NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(friend void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { nlohmann::to_json(nlohmann_json_j, static_cast(nlohmann_json_t)); }) \ + /* NOLINTNEXTLINE(bugprone-macro-parentheses) Type/BaseType are used as declarator types, not in expressions */ \ + NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(friend void from_json(const BasicJsonType& nlohmann_json_j, Type& nlohmann_json_t) { nlohmann::from_json(nlohmann_json_j, static_cast(nlohmann_json_t)); }) + +#define NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE(...) NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_DERIVED_TYPE_BODY(NLOHMANN_JSON_DEFINE_DERIVED_TYPE_INTRUSIVE_, __VA_ARGS__)(__VA_ARGS__)) + #define NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE_WITH_NAMES(Type, BaseType, ...) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(friend void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { nlohmann::to_json(nlohmann_json_j, static_cast(nlohmann_json_t)); NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_DOUBLE_PASTE(NLOHMANN_JSON_TO_WITH_NAME, __VA_ARGS__)) }) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(friend void from_json(const BasicJsonType& nlohmann_json_j, Type& nlohmann_json_t) { nlohmann::from_json(nlohmann_json_j, static_cast(nlohmann_json_t)); NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_DOUBLE_PASTE(NLOHMANN_JSON_FROM_WITH_NAME, __VA_ARGS__)) }) @@ -696,10 +776,15 @@ void templated_json_throw(ExceptionType exception) @since version 3.12.0 @sa https://json.nlohmann.me/api/macros/nlohmann_define_derived_type/ */ -#define NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE_WITH_DEFAULT(Type, BaseType, ...) \ +#define NLOHMANN_JSON_DEFINE_DERIVED_TYPE_INTRUSIVE_WITH_DEFAULT_MEMBERS(Type, BaseType, ...) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(friend void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { nlohmann::to_json(nlohmann_json_j, static_cast(nlohmann_json_t)); NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_PASTE(NLOHMANN_JSON_TO, __VA_ARGS__)) }) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(friend void from_json(const BasicJsonType& nlohmann_json_j, Type& nlohmann_json_t) { nlohmann::from_json(nlohmann_json_j, static_cast(nlohmann_json_t)); const Type nlohmann_json_default_obj{}; NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_PASTE(NLOHMANN_JSON_FROM_WITH_DEFAULT, __VA_ARGS__)) }) +// identical to NLOHMANN_JSON_DEFINE_DERIVED_TYPE_INTRUSIVE_EMPTY: with no members there is nothing to default +#define NLOHMANN_JSON_DEFINE_DERIVED_TYPE_INTRUSIVE_WITH_DEFAULT_EMPTY(Type, BaseType) NLOHMANN_JSON_DEFINE_DERIVED_TYPE_INTRUSIVE_EMPTY(Type, BaseType) + +#define NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE_WITH_DEFAULT(...) NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_DERIVED_TYPE_BODY(NLOHMANN_JSON_DEFINE_DERIVED_TYPE_INTRUSIVE_WITH_DEFAULT_, __VA_ARGS__)(__VA_ARGS__)) + #define NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE_WITH_DEFAULT_WITH_NAMES(Type, BaseType, ...) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(friend void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { nlohmann::to_json(nlohmann_json_j, static_cast(nlohmann_json_t)); NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_DOUBLE_PASTE(NLOHMANN_JSON_TO_WITH_NAME, __VA_ARGS__)) }) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(friend void from_json(const BasicJsonType& nlohmann_json_j, Type& nlohmann_json_t) { nlohmann::from_json(nlohmann_json_j, static_cast(nlohmann_json_t)); const Type nlohmann_json_default_obj{}; NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_DOUBLE_PASTE(NLOHMANN_JSON_FROM_WITH_DEFAULT_WITH_NAME, __VA_ARGS__)) }) @@ -710,9 +795,14 @@ void templated_json_throw(ExceptionType exception) @since version 3.12.0 @sa https://json.nlohmann.me/api/macros/nlohmann_define_derived_type/ */ -#define NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE_ONLY_SERIALIZE(Type, BaseType, ...) \ +#define NLOHMANN_JSON_DEFINE_DERIVED_TYPE_INTRUSIVE_ONLY_SERIALIZE_MEMBERS(Type, BaseType, ...) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(friend void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { nlohmann::to_json(nlohmann_json_j, static_cast(nlohmann_json_t)); NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_PASTE(NLOHMANN_JSON_TO, __VA_ARGS__)) }) +#define NLOHMANN_JSON_DEFINE_DERIVED_TYPE_INTRUSIVE_ONLY_SERIALIZE_EMPTY(Type, BaseType) \ + NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(friend void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { nlohmann::to_json(nlohmann_json_j, static_cast(nlohmann_json_t)); }) + +#define NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE_ONLY_SERIALIZE(...) NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_DERIVED_TYPE_BODY(NLOHMANN_JSON_DEFINE_DERIVED_TYPE_INTRUSIVE_ONLY_SERIALIZE_, __VA_ARGS__)(__VA_ARGS__)) + #define NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE_ONLY_SERIALIZE_WITH_NAMES(Type, BaseType, ...) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(friend void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { nlohmann::to_json(nlohmann_json_j, static_cast(nlohmann_json_t)); NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_DOUBLE_PASTE(NLOHMANN_JSON_TO_WITH_NAME, __VA_ARGS__)) }) @@ -723,10 +813,17 @@ void templated_json_throw(ExceptionType exception) @since version 3.12.0 @sa https://json.nlohmann.me/api/macros/nlohmann_define_derived_type/ */ -#define NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE(Type, BaseType, ...) \ +#define NLOHMANN_JSON_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_MEMBERS(Type, BaseType, ...) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { nlohmann::to_json(nlohmann_json_j, static_cast(nlohmann_json_t)); NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_PASTE(NLOHMANN_JSON_TO, __VA_ARGS__)) }) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(void from_json(const BasicJsonType& nlohmann_json_j, Type& nlohmann_json_t) { nlohmann::from_json(nlohmann_json_j, static_cast(nlohmann_json_t)); NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_PASTE(NLOHMANN_JSON_FROM, __VA_ARGS__)) }) +#define NLOHMANN_JSON_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_EMPTY(Type, BaseType) \ + NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { nlohmann::to_json(nlohmann_json_j, static_cast(nlohmann_json_t)); }) \ + /* NOLINTNEXTLINE(bugprone-macro-parentheses) Type/BaseType are used as declarator types, not in expressions */ \ + NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(void from_json(const BasicJsonType& nlohmann_json_j, Type& nlohmann_json_t) { nlohmann::from_json(nlohmann_json_j, static_cast(nlohmann_json_t)); }) + +#define NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE(...) NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_DERIVED_TYPE_BODY(NLOHMANN_JSON_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_, __VA_ARGS__)(__VA_ARGS__)) + #define NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_WITH_NAMES(Type, BaseType, ...) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { nlohmann::to_json(nlohmann_json_j, static_cast(nlohmann_json_t)); NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_DOUBLE_PASTE(NLOHMANN_JSON_TO_WITH_NAME, __VA_ARGS__)) }) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(void from_json(const BasicJsonType& nlohmann_json_j, Type& nlohmann_json_t) { nlohmann::from_json(nlohmann_json_j, static_cast(nlohmann_json_t)); NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_DOUBLE_PASTE(NLOHMANN_JSON_FROM_WITH_NAME, __VA_ARGS__)) }) @@ -737,10 +834,15 @@ void templated_json_throw(ExceptionType exception) @since version 3.12.0 @sa https://json.nlohmann.me/api/macros/nlohmann_define_derived_type/ */ -#define NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_WITH_DEFAULT(Type, BaseType, ...) \ +#define NLOHMANN_JSON_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_WITH_DEFAULT_MEMBERS(Type, BaseType, ...) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { nlohmann::to_json(nlohmann_json_j, static_cast(nlohmann_json_t)); NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_PASTE(NLOHMANN_JSON_TO, __VA_ARGS__)) }) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(void from_json(const BasicJsonType& nlohmann_json_j, Type& nlohmann_json_t) { nlohmann::from_json(nlohmann_json_j, static_cast(nlohmann_json_t)); const Type nlohmann_json_default_obj{}; NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_PASTE(NLOHMANN_JSON_FROM_WITH_DEFAULT, __VA_ARGS__)) }) +// identical to NLOHMANN_JSON_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_EMPTY: with no members there is nothing to default +#define NLOHMANN_JSON_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_WITH_DEFAULT_EMPTY(Type, BaseType) NLOHMANN_JSON_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_EMPTY(Type, BaseType) + +#define NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_WITH_DEFAULT(...) NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_DERIVED_TYPE_BODY(NLOHMANN_JSON_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_WITH_DEFAULT_, __VA_ARGS__)(__VA_ARGS__)) + #define NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_WITH_DEFAULT_WITH_NAMES(Type, BaseType, ...) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { nlohmann::to_json(nlohmann_json_j, static_cast(nlohmann_json_t)); NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_DOUBLE_PASTE(NLOHMANN_JSON_TO_WITH_NAME, __VA_ARGS__)) }) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(void from_json(const BasicJsonType& nlohmann_json_j, Type& nlohmann_json_t) { nlohmann::from_json(nlohmann_json_j, static_cast(nlohmann_json_t)); const Type nlohmann_json_default_obj{}; NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_DOUBLE_PASTE(NLOHMANN_JSON_FROM_WITH_DEFAULT_WITH_NAME, __VA_ARGS__)) }) @@ -751,9 +853,14 @@ void templated_json_throw(ExceptionType exception) @since version 3.12.0 @sa https://json.nlohmann.me/api/macros/nlohmann_define_derived_type/ */ -#define NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_ONLY_SERIALIZE(Type, BaseType, ...) \ +#define NLOHMANN_JSON_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_ONLY_SERIALIZE_MEMBERS(Type, BaseType, ...) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { nlohmann::to_json(nlohmann_json_j, static_cast(nlohmann_json_t)); NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_PASTE(NLOHMANN_JSON_TO, __VA_ARGS__)) }) +#define NLOHMANN_JSON_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_ONLY_SERIALIZE_EMPTY(Type, BaseType) \ + NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { nlohmann::to_json(nlohmann_json_j, static_cast(nlohmann_json_t)); }) + +#define NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_ONLY_SERIALIZE(...) NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_DERIVED_TYPE_BODY(NLOHMANN_JSON_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_ONLY_SERIALIZE_, __VA_ARGS__)(__VA_ARGS__)) + #define NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_ONLY_SERIALIZE_WITH_NAMES(Type, BaseType, ...) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { nlohmann::to_json(nlohmann_json_j, static_cast(nlohmann_json_t)); NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_DOUBLE_PASTE(NLOHMANN_JSON_TO_WITH_NAME, __VA_ARGS__)) }) diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 8c3a1319b..e6fe169ef 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -2996,16 +2996,62 @@ void templated_json_throw(ExceptionType exception) #define NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(...) template::value, int> = 0> __VA_ARGS__ +// Helpers used to dispatch the NLOHMANN_DEFINE_TYPE_*/NLOHMANN_DEFINE_DERIVED_TYPE_* +// macros below between a zero-member and a one-or-more-member implementation +// (issue #4041, e.g. NLOHMANN_DEFINE_TYPE_INTRUSIVE(Type) with no further +// arguments). NLOHMANN_JSON_TYPE_BODY(Prefix, ...) expands to the macro name +// Prefix##EMPTY when __VA_ARGS__ is a single argument (Type alone) and to +// Prefix##MEMBERS for two or more (Type, member...). It reuses the existing +// 64-slot NLOHMANN_JSON_GET_MACRO dispatch with one extra trailing sentinel +// token appended so its own trailing "..." is never left completely empty at +// the lowest supported argument count -- invoking a variadic macro so that +// "..." matches nothing is only granted unconditionally by the standard since +// C++20, and pre-C++20 compilers may reject it under -pedantic regardless of +// what the macro body does. +// +// The EMPTY/MEMBERS suffixes are pasted onto Prefix right in the slot table: +// operands of ## are not macro-expanded, so the dispatch keeps working even if +// user code defines macros named EMPTY or MEMBERS. Producing the bare suffix +// first and pasting it later would let such a macro replace it. +// +// NLOHMANN_JSON_DERIVED_TYPE_BODY(Prefix, ...) answers the same question for +// the derived-type macros, whose fixed prefix is Type,BaseType. It drops the +// leading Type and defers to NLOHMANN_JSON_TYPE_BODY rather than shifting the +// slot table by one: NLOHMANN_JSON_GET_MACRO only resolves 64 positional +// arguments, so dispatching on Type,BaseType,member... directly would run out +// one slot early and cap the derived-type macros at 62 members instead of the +// 63 that NLOHMANN_JSON_PASTE supports. +#define NLOHMANN_JSON_TYPE_BODY(Prefix, ...) NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_GET_MACRO(__VA_ARGS__, \ + Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, \ + Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, \ + Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, \ + Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, \ + Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, \ + Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, \ + Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, \ + Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## MEMBERS, Prefix ## EMPTY, \ + NLOHMANN_JSON_TYPE_BODY_SENTINEL)) + +#define NLOHMANN_JSON_DERIVED_TYPE_BODY_(Prefix, Type, ...) NLOHMANN_JSON_TYPE_BODY(Prefix, __VA_ARGS__) +#define NLOHMANN_JSON_DERIVED_TYPE_BODY(Prefix, ...) NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_DERIVED_TYPE_BODY_(Prefix, __VA_ARGS__)) + /*! @brief macro @def NLOHMANN_DEFINE_TYPE_INTRUSIVE @since version 3.9.0 @sa https://json.nlohmann.me/api/macros/nlohmann_define_type_intrusive/ */ -#define NLOHMANN_DEFINE_TYPE_INTRUSIVE(Type, ...) \ +#define NLOHMANN_JSON_DEFINE_TYPE_INTRUSIVE_MEMBERS(Type, ...) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(friend void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_PASTE(NLOHMANN_JSON_TO, __VA_ARGS__)) }) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(friend void from_json(const BasicJsonType& nlohmann_json_j, Type& nlohmann_json_t) { NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_PASTE(NLOHMANN_JSON_FROM, __VA_ARGS__)) }) +#define NLOHMANN_JSON_DEFINE_TYPE_INTRUSIVE_EMPTY(Type) \ + NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(friend void to_json(BasicJsonType& nlohmann_json_j, const Type&) { nlohmann_json_j = BasicJsonType::object(); }) \ + /* NOLINTNEXTLINE(bugprone-macro-parentheses) Type is used as a declarator type, not in an expression */ \ + NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(friend void from_json(const BasicJsonType&, Type&) noexcept { }) + +#define NLOHMANN_DEFINE_TYPE_INTRUSIVE(...) NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_TYPE_BODY(NLOHMANN_JSON_DEFINE_TYPE_INTRUSIVE_, __VA_ARGS__)(__VA_ARGS__)) + #define NLOHMANN_DEFINE_TYPE_INTRUSIVE_WITH_NAMES(Type, ...) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(friend void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_DOUBLE_PASTE(NLOHMANN_JSON_TO_WITH_NAME, __VA_ARGS__)) }) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(friend void from_json(const BasicJsonType& nlohmann_json_j, Type& nlohmann_json_t) { NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_DOUBLE_PASTE(NLOHMANN_JSON_FROM_WITH_NAME, __VA_ARGS__)) }) @@ -3016,10 +3062,15 @@ void templated_json_throw(ExceptionType exception) @since version 3.11.0 @sa https://json.nlohmann.me/api/macros/nlohmann_define_type_intrusive/ */ -#define NLOHMANN_DEFINE_TYPE_INTRUSIVE_WITH_DEFAULT(Type, ...) \ +#define NLOHMANN_JSON_DEFINE_TYPE_INTRUSIVE_WITH_DEFAULT_MEMBERS(Type, ...) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(friend void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_PASTE(NLOHMANN_JSON_TO, __VA_ARGS__)) }) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(friend void from_json(const BasicJsonType& nlohmann_json_j, Type& nlohmann_json_t) { const Type nlohmann_json_default_obj{}; NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_PASTE(NLOHMANN_JSON_FROM_WITH_DEFAULT, __VA_ARGS__)) }) +// identical to NLOHMANN_JSON_DEFINE_TYPE_INTRUSIVE_EMPTY: with no members there is nothing to default +#define NLOHMANN_JSON_DEFINE_TYPE_INTRUSIVE_WITH_DEFAULT_EMPTY(Type) NLOHMANN_JSON_DEFINE_TYPE_INTRUSIVE_EMPTY(Type) + +#define NLOHMANN_DEFINE_TYPE_INTRUSIVE_WITH_DEFAULT(...) NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_TYPE_BODY(NLOHMANN_JSON_DEFINE_TYPE_INTRUSIVE_WITH_DEFAULT_, __VA_ARGS__)(__VA_ARGS__)) + #define NLOHMANN_DEFINE_TYPE_INTRUSIVE_WITH_DEFAULT_WITH_NAMES(Type, ...) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(friend void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_DOUBLE_PASTE(NLOHMANN_JSON_TO_WITH_NAME, __VA_ARGS__)) }) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(friend void from_json(const BasicJsonType& nlohmann_json_j, Type& nlohmann_json_t) { const Type nlohmann_json_default_obj{}; NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_DOUBLE_PASTE(NLOHMANN_JSON_FROM_WITH_DEFAULT_WITH_NAME, __VA_ARGS__)) }) @@ -3030,9 +3081,14 @@ void templated_json_throw(ExceptionType exception) @since version 3.11.3 @sa https://json.nlohmann.me/api/macros/nlohmann_define_type_intrusive/ */ -#define NLOHMANN_DEFINE_TYPE_INTRUSIVE_ONLY_SERIALIZE(Type, ...) \ +#define NLOHMANN_JSON_DEFINE_TYPE_INTRUSIVE_ONLY_SERIALIZE_MEMBERS(Type, ...) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(friend void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_PASTE(NLOHMANN_JSON_TO, __VA_ARGS__)) }) +#define NLOHMANN_JSON_DEFINE_TYPE_INTRUSIVE_ONLY_SERIALIZE_EMPTY(Type) \ + NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(friend void to_json(BasicJsonType& nlohmann_json_j, const Type&) { nlohmann_json_j = BasicJsonType::object(); }) + +#define NLOHMANN_DEFINE_TYPE_INTRUSIVE_ONLY_SERIALIZE(...) NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_TYPE_BODY(NLOHMANN_JSON_DEFINE_TYPE_INTRUSIVE_ONLY_SERIALIZE_, __VA_ARGS__)(__VA_ARGS__)) + #define NLOHMANN_DEFINE_TYPE_INTRUSIVE_ONLY_SERIALIZE_WITH_NAMES(Type, ...) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(friend void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_DOUBLE_PASTE(NLOHMANN_JSON_TO_WITH_NAME, __VA_ARGS__)) }) @@ -3042,10 +3098,17 @@ void templated_json_throw(ExceptionType exception) @since version 3.9.0 @sa https://json.nlohmann.me/api/macros/nlohmann_define_type_non_intrusive/ */ -#define NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE(Type, ...) \ +#define NLOHMANN_JSON_DEFINE_TYPE_NON_INTRUSIVE_MEMBERS(Type, ...) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_PASTE(NLOHMANN_JSON_TO, __VA_ARGS__)) }) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(void from_json(const BasicJsonType& nlohmann_json_j, Type& nlohmann_json_t) { NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_PASTE(NLOHMANN_JSON_FROM, __VA_ARGS__)) }) +#define NLOHMANN_JSON_DEFINE_TYPE_NON_INTRUSIVE_EMPTY(Type) \ + NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(void to_json(BasicJsonType& nlohmann_json_j, const Type&) { nlohmann_json_j = BasicJsonType::object(); }) \ + /* NOLINTNEXTLINE(bugprone-macro-parentheses) Type is used as a declarator type, not in an expression */ \ + NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(void from_json(const BasicJsonType&, Type&) noexcept { }) + +#define NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE(...) NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_TYPE_BODY(NLOHMANN_JSON_DEFINE_TYPE_NON_INTRUSIVE_, __VA_ARGS__)(__VA_ARGS__)) + #define NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE_WITH_NAMES(Type, ...) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_DOUBLE_PASTE(NLOHMANN_JSON_TO_WITH_NAME, __VA_ARGS__)) }) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(void from_json(const BasicJsonType& nlohmann_json_j, Type& nlohmann_json_t) { NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_DOUBLE_PASTE(NLOHMANN_JSON_FROM_WITH_NAME, __VA_ARGS__)) }) @@ -3056,10 +3119,15 @@ void templated_json_throw(ExceptionType exception) @since version 3.11.0 @sa https://json.nlohmann.me/api/macros/nlohmann_define_type_non_intrusive/ */ -#define NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE_WITH_DEFAULT(Type, ...) \ +#define NLOHMANN_JSON_DEFINE_TYPE_NON_INTRUSIVE_WITH_DEFAULT_MEMBERS(Type, ...) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_PASTE(NLOHMANN_JSON_TO, __VA_ARGS__)) }) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(void from_json(const BasicJsonType& nlohmann_json_j, Type& nlohmann_json_t) { const Type nlohmann_json_default_obj{}; NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_PASTE(NLOHMANN_JSON_FROM_WITH_DEFAULT, __VA_ARGS__)) }) +// identical to NLOHMANN_JSON_DEFINE_TYPE_NON_INTRUSIVE_EMPTY: with no members there is nothing to default +#define NLOHMANN_JSON_DEFINE_TYPE_NON_INTRUSIVE_WITH_DEFAULT_EMPTY(Type) NLOHMANN_JSON_DEFINE_TYPE_NON_INTRUSIVE_EMPTY(Type) + +#define NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE_WITH_DEFAULT(...) NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_TYPE_BODY(NLOHMANN_JSON_DEFINE_TYPE_NON_INTRUSIVE_WITH_DEFAULT_, __VA_ARGS__)(__VA_ARGS__)) + #define NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE_WITH_DEFAULT_WITH_NAMES(Type, ...) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_DOUBLE_PASTE(NLOHMANN_JSON_TO_WITH_NAME, __VA_ARGS__)) }) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(void from_json(const BasicJsonType& nlohmann_json_j, Type& nlohmann_json_t) { const Type nlohmann_json_default_obj{}; NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_DOUBLE_PASTE(NLOHMANN_JSON_FROM_WITH_DEFAULT_WITH_NAME, __VA_ARGS__)) }) @@ -3070,9 +3138,14 @@ void templated_json_throw(ExceptionType exception) @since version 3.11.3 @sa https://json.nlohmann.me/api/macros/nlohmann_define_type_non_intrusive/ */ -#define NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE_ONLY_SERIALIZE(Type, ...) \ +#define NLOHMANN_JSON_DEFINE_TYPE_NON_INTRUSIVE_ONLY_SERIALIZE_MEMBERS(Type, ...) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_PASTE(NLOHMANN_JSON_TO, __VA_ARGS__)) }) +#define NLOHMANN_JSON_DEFINE_TYPE_NON_INTRUSIVE_ONLY_SERIALIZE_EMPTY(Type) \ + NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(void to_json(BasicJsonType& nlohmann_json_j, const Type&) { nlohmann_json_j = BasicJsonType::object(); }) + +#define NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE_ONLY_SERIALIZE(...) NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_TYPE_BODY(NLOHMANN_JSON_DEFINE_TYPE_NON_INTRUSIVE_ONLY_SERIALIZE_, __VA_ARGS__)(__VA_ARGS__)) + #define NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE_ONLY_SERIALIZE_WITH_NAMES(Type, ...) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_DOUBLE_PASTE(NLOHMANN_JSON_TO_WITH_NAME, __VA_ARGS__)) }) @@ -3082,10 +3155,17 @@ void templated_json_throw(ExceptionType exception) @since version 3.12.0 @sa https://json.nlohmann.me/api/macros/nlohmann_define_derived_type/ */ -#define NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE(Type, BaseType, ...) \ +#define NLOHMANN_JSON_DEFINE_DERIVED_TYPE_INTRUSIVE_MEMBERS(Type, BaseType, ...) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(friend void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { nlohmann::to_json(nlohmann_json_j, static_cast(nlohmann_json_t)); NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_PASTE(NLOHMANN_JSON_TO, __VA_ARGS__)) }) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(friend void from_json(const BasicJsonType& nlohmann_json_j, Type& nlohmann_json_t) { nlohmann::from_json(nlohmann_json_j, static_cast(nlohmann_json_t)); NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_PASTE(NLOHMANN_JSON_FROM, __VA_ARGS__)) }) +#define NLOHMANN_JSON_DEFINE_DERIVED_TYPE_INTRUSIVE_EMPTY(Type, BaseType) \ + NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(friend void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { nlohmann::to_json(nlohmann_json_j, static_cast(nlohmann_json_t)); }) \ + /* NOLINTNEXTLINE(bugprone-macro-parentheses) Type/BaseType are used as declarator types, not in expressions */ \ + NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(friend void from_json(const BasicJsonType& nlohmann_json_j, Type& nlohmann_json_t) { nlohmann::from_json(nlohmann_json_j, static_cast(nlohmann_json_t)); }) + +#define NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE(...) NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_DERIVED_TYPE_BODY(NLOHMANN_JSON_DEFINE_DERIVED_TYPE_INTRUSIVE_, __VA_ARGS__)(__VA_ARGS__)) + #define NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE_WITH_NAMES(Type, BaseType, ...) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(friend void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { nlohmann::to_json(nlohmann_json_j, static_cast(nlohmann_json_t)); NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_DOUBLE_PASTE(NLOHMANN_JSON_TO_WITH_NAME, __VA_ARGS__)) }) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(friend void from_json(const BasicJsonType& nlohmann_json_j, Type& nlohmann_json_t) { nlohmann::from_json(nlohmann_json_j, static_cast(nlohmann_json_t)); NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_DOUBLE_PASTE(NLOHMANN_JSON_FROM_WITH_NAME, __VA_ARGS__)) }) @@ -3096,10 +3176,15 @@ void templated_json_throw(ExceptionType exception) @since version 3.12.0 @sa https://json.nlohmann.me/api/macros/nlohmann_define_derived_type/ */ -#define NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE_WITH_DEFAULT(Type, BaseType, ...) \ +#define NLOHMANN_JSON_DEFINE_DERIVED_TYPE_INTRUSIVE_WITH_DEFAULT_MEMBERS(Type, BaseType, ...) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(friend void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { nlohmann::to_json(nlohmann_json_j, static_cast(nlohmann_json_t)); NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_PASTE(NLOHMANN_JSON_TO, __VA_ARGS__)) }) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(friend void from_json(const BasicJsonType& nlohmann_json_j, Type& nlohmann_json_t) { nlohmann::from_json(nlohmann_json_j, static_cast(nlohmann_json_t)); const Type nlohmann_json_default_obj{}; NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_PASTE(NLOHMANN_JSON_FROM_WITH_DEFAULT, __VA_ARGS__)) }) +// identical to NLOHMANN_JSON_DEFINE_DERIVED_TYPE_INTRUSIVE_EMPTY: with no members there is nothing to default +#define NLOHMANN_JSON_DEFINE_DERIVED_TYPE_INTRUSIVE_WITH_DEFAULT_EMPTY(Type, BaseType) NLOHMANN_JSON_DEFINE_DERIVED_TYPE_INTRUSIVE_EMPTY(Type, BaseType) + +#define NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE_WITH_DEFAULT(...) NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_DERIVED_TYPE_BODY(NLOHMANN_JSON_DEFINE_DERIVED_TYPE_INTRUSIVE_WITH_DEFAULT_, __VA_ARGS__)(__VA_ARGS__)) + #define NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE_WITH_DEFAULT_WITH_NAMES(Type, BaseType, ...) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(friend void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { nlohmann::to_json(nlohmann_json_j, static_cast(nlohmann_json_t)); NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_DOUBLE_PASTE(NLOHMANN_JSON_TO_WITH_NAME, __VA_ARGS__)) }) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(friend void from_json(const BasicJsonType& nlohmann_json_j, Type& nlohmann_json_t) { nlohmann::from_json(nlohmann_json_j, static_cast(nlohmann_json_t)); const Type nlohmann_json_default_obj{}; NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_DOUBLE_PASTE(NLOHMANN_JSON_FROM_WITH_DEFAULT_WITH_NAME, __VA_ARGS__)) }) @@ -3110,9 +3195,14 @@ void templated_json_throw(ExceptionType exception) @since version 3.12.0 @sa https://json.nlohmann.me/api/macros/nlohmann_define_derived_type/ */ -#define NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE_ONLY_SERIALIZE(Type, BaseType, ...) \ +#define NLOHMANN_JSON_DEFINE_DERIVED_TYPE_INTRUSIVE_ONLY_SERIALIZE_MEMBERS(Type, BaseType, ...) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(friend void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { nlohmann::to_json(nlohmann_json_j, static_cast(nlohmann_json_t)); NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_PASTE(NLOHMANN_JSON_TO, __VA_ARGS__)) }) +#define NLOHMANN_JSON_DEFINE_DERIVED_TYPE_INTRUSIVE_ONLY_SERIALIZE_EMPTY(Type, BaseType) \ + NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(friend void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { nlohmann::to_json(nlohmann_json_j, static_cast(nlohmann_json_t)); }) + +#define NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE_ONLY_SERIALIZE(...) NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_DERIVED_TYPE_BODY(NLOHMANN_JSON_DEFINE_DERIVED_TYPE_INTRUSIVE_ONLY_SERIALIZE_, __VA_ARGS__)(__VA_ARGS__)) + #define NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE_ONLY_SERIALIZE_WITH_NAMES(Type, BaseType, ...) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(friend void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { nlohmann::to_json(nlohmann_json_j, static_cast(nlohmann_json_t)); NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_DOUBLE_PASTE(NLOHMANN_JSON_TO_WITH_NAME, __VA_ARGS__)) }) @@ -3123,10 +3213,17 @@ void templated_json_throw(ExceptionType exception) @since version 3.12.0 @sa https://json.nlohmann.me/api/macros/nlohmann_define_derived_type/ */ -#define NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE(Type, BaseType, ...) \ +#define NLOHMANN_JSON_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_MEMBERS(Type, BaseType, ...) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { nlohmann::to_json(nlohmann_json_j, static_cast(nlohmann_json_t)); NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_PASTE(NLOHMANN_JSON_TO, __VA_ARGS__)) }) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(void from_json(const BasicJsonType& nlohmann_json_j, Type& nlohmann_json_t) { nlohmann::from_json(nlohmann_json_j, static_cast(nlohmann_json_t)); NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_PASTE(NLOHMANN_JSON_FROM, __VA_ARGS__)) }) +#define NLOHMANN_JSON_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_EMPTY(Type, BaseType) \ + NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { nlohmann::to_json(nlohmann_json_j, static_cast(nlohmann_json_t)); }) \ + /* NOLINTNEXTLINE(bugprone-macro-parentheses) Type/BaseType are used as declarator types, not in expressions */ \ + NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(void from_json(const BasicJsonType& nlohmann_json_j, Type& nlohmann_json_t) { nlohmann::from_json(nlohmann_json_j, static_cast(nlohmann_json_t)); }) + +#define NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE(...) NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_DERIVED_TYPE_BODY(NLOHMANN_JSON_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_, __VA_ARGS__)(__VA_ARGS__)) + #define NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_WITH_NAMES(Type, BaseType, ...) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { nlohmann::to_json(nlohmann_json_j, static_cast(nlohmann_json_t)); NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_DOUBLE_PASTE(NLOHMANN_JSON_TO_WITH_NAME, __VA_ARGS__)) }) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(void from_json(const BasicJsonType& nlohmann_json_j, Type& nlohmann_json_t) { nlohmann::from_json(nlohmann_json_j, static_cast(nlohmann_json_t)); NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_DOUBLE_PASTE(NLOHMANN_JSON_FROM_WITH_NAME, __VA_ARGS__)) }) @@ -3137,10 +3234,15 @@ void templated_json_throw(ExceptionType exception) @since version 3.12.0 @sa https://json.nlohmann.me/api/macros/nlohmann_define_derived_type/ */ -#define NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_WITH_DEFAULT(Type, BaseType, ...) \ +#define NLOHMANN_JSON_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_WITH_DEFAULT_MEMBERS(Type, BaseType, ...) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { nlohmann::to_json(nlohmann_json_j, static_cast(nlohmann_json_t)); NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_PASTE(NLOHMANN_JSON_TO, __VA_ARGS__)) }) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(void from_json(const BasicJsonType& nlohmann_json_j, Type& nlohmann_json_t) { nlohmann::from_json(nlohmann_json_j, static_cast(nlohmann_json_t)); const Type nlohmann_json_default_obj{}; NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_PASTE(NLOHMANN_JSON_FROM_WITH_DEFAULT, __VA_ARGS__)) }) +// identical to NLOHMANN_JSON_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_EMPTY: with no members there is nothing to default +#define NLOHMANN_JSON_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_WITH_DEFAULT_EMPTY(Type, BaseType) NLOHMANN_JSON_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_EMPTY(Type, BaseType) + +#define NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_WITH_DEFAULT(...) NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_DERIVED_TYPE_BODY(NLOHMANN_JSON_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_WITH_DEFAULT_, __VA_ARGS__)(__VA_ARGS__)) + #define NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_WITH_DEFAULT_WITH_NAMES(Type, BaseType, ...) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { nlohmann::to_json(nlohmann_json_j, static_cast(nlohmann_json_t)); NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_DOUBLE_PASTE(NLOHMANN_JSON_TO_WITH_NAME, __VA_ARGS__)) }) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(void from_json(const BasicJsonType& nlohmann_json_j, Type& nlohmann_json_t) { nlohmann::from_json(nlohmann_json_j, static_cast(nlohmann_json_t)); const Type nlohmann_json_default_obj{}; NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_DOUBLE_PASTE(NLOHMANN_JSON_FROM_WITH_DEFAULT_WITH_NAME, __VA_ARGS__)) }) @@ -3151,9 +3253,14 @@ void templated_json_throw(ExceptionType exception) @since version 3.12.0 @sa https://json.nlohmann.me/api/macros/nlohmann_define_derived_type/ */ -#define NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_ONLY_SERIALIZE(Type, BaseType, ...) \ +#define NLOHMANN_JSON_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_ONLY_SERIALIZE_MEMBERS(Type, BaseType, ...) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { nlohmann::to_json(nlohmann_json_j, static_cast(nlohmann_json_t)); NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_PASTE(NLOHMANN_JSON_TO, __VA_ARGS__)) }) +#define NLOHMANN_JSON_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_ONLY_SERIALIZE_EMPTY(Type, BaseType) \ + NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { nlohmann::to_json(nlohmann_json_j, static_cast(nlohmann_json_t)); }) + +#define NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_ONLY_SERIALIZE(...) NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_DERIVED_TYPE_BODY(NLOHMANN_JSON_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_ONLY_SERIALIZE_, __VA_ARGS__)(__VA_ARGS__)) + #define NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_ONLY_SERIALIZE_WITH_NAMES(Type, BaseType, ...) \ NLOHMANN_JSON_BASIC_TYPE_TEMPLATE(void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { nlohmann::to_json(nlohmann_json_j, static_cast(nlohmann_json_t)); NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_DOUBLE_PASTE(NLOHMANN_JSON_TO_WITH_NAME, __VA_ARGS__)) }) diff --git a/tests/src/unit-udt_macro.cpp b/tests/src/unit-udt_macro.cpp index 7fab80921..ff254b60e 100644 --- a/tests/src/unit-udt_macro.cpp +++ b/tests/src/unit-udt_macro.cpp @@ -778,6 +778,193 @@ class derived_person_only_serialize_private_3 : person_without_default_construct NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE_ONLY_SERIALIZE_WITH_NAMES(derived_person_only_serialize_private_3, person_without_default_constructor_3, "json_hair_color", hair_color) }; +// Zero-member types for issue #4041: NLOHMANN_DEFINE_TYPE_* and +// NLOHMANN_DEFINE_DERIVED_TYPE_* must compile and produce a valid (empty) +// JSON object when no member arguments are given. +class empty_intrusive +{ + public: + bool operator==(const empty_intrusive& /*rhs*/) const + { + return true; + } + NLOHMANN_DEFINE_TYPE_INTRUSIVE(empty_intrusive) +}; + +class empty_intrusive_with_default +{ + public: + bool operator==(const empty_intrusive_with_default& /*rhs*/) const + { + return true; + } + NLOHMANN_DEFINE_TYPE_INTRUSIVE_WITH_DEFAULT(empty_intrusive_with_default) +}; + +class empty_intrusive_only_serialize +{ + public: + NLOHMANN_DEFINE_TYPE_INTRUSIVE_ONLY_SERIALIZE(empty_intrusive_only_serialize) +}; + +class empty_non_intrusive +{ + public: + bool operator==(const empty_non_intrusive& /*rhs*/) const + { + return true; + } +}; +// NOLINTNEXTLINE(misc-use-internal-linkage) +NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE(empty_non_intrusive) + +class empty_non_intrusive_with_default +{ + public: + bool operator==(const empty_non_intrusive_with_default& /*rhs*/) const + { + return true; + } +}; +// NOLINTNEXTLINE(misc-use-internal-linkage) +NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE_WITH_DEFAULT(empty_non_intrusive_with_default) + +class empty_non_intrusive_only_serialize {}; +// NOLINTNEXTLINE(misc-use-internal-linkage) +NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE_ONLY_SERIALIZE(empty_non_intrusive_only_serialize) + +class empty_derived_intrusive : public person_with_private_data +{ + public: + empty_derived_intrusive() = default; + empty_derived_intrusive(std::string name_, int age_, json metadata_) + : person_with_private_data(std::move(name_), age_, std::move(metadata_)) + {} + NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE(empty_derived_intrusive, person_with_private_data) +}; + +class empty_derived_intrusive_with_default : public person_with_private_data +{ + public: + empty_derived_intrusive_with_default() = default; + empty_derived_intrusive_with_default(std::string name_, int age_, json metadata_) + : person_with_private_data(std::move(name_), age_, std::move(metadata_)) + {} + NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE_WITH_DEFAULT(empty_derived_intrusive_with_default, person_with_private_data) +}; + +class empty_derived_intrusive_only_serialize : public person_with_private_data +{ + public: + empty_derived_intrusive_only_serialize() = default; + empty_derived_intrusive_only_serialize(std::string name_, int age_, json metadata_) + : person_with_private_data(std::move(name_), age_, std::move(metadata_)) + {} + NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE_ONLY_SERIALIZE(empty_derived_intrusive_only_serialize, person_with_private_data) +}; + +class empty_derived_non_intrusive : public person_with_private_data +{ + public: + empty_derived_non_intrusive() = default; + empty_derived_non_intrusive(std::string name_, int age_, json metadata_) + : person_with_private_data(std::move(name_), age_, std::move(metadata_)) + {} +}; +// NOLINTNEXTLINE(misc-use-internal-linkage) +NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE(empty_derived_non_intrusive, person_with_private_data) + +class empty_derived_non_intrusive_with_default : public person_with_private_data +{ + public: + empty_derived_non_intrusive_with_default() = default; + empty_derived_non_intrusive_with_default(std::string name_, int age_, json metadata_) + : person_with_private_data(std::move(name_), age_, std::move(metadata_)) + {} +}; +// NOLINTNEXTLINE(misc-use-internal-linkage) +NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_WITH_DEFAULT(empty_derived_non_intrusive_with_default, person_with_private_data) + +class empty_derived_non_intrusive_only_serialize : public person_with_private_data +{ + public: + empty_derived_non_intrusive_only_serialize() = default; + empty_derived_non_intrusive_only_serialize(std::string name_, int age_, json metadata_) + : person_with_private_data(std::move(name_), age_, std::move(metadata_)) + {} +}; +// NOLINTNEXTLINE(misc-use-internal-linkage) +NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_ONLY_SERIALIZE(empty_derived_non_intrusive_only_serialize, person_with_private_data) + +// Types at the documented maximum member count (63) for issue #4041's +// argument-count dispatch. The derived-type macros carry a two-token +// Type,BaseType prefix, so they reach two slots further into +// NLOHMANN_JSON_GET_MACRO than the non-derived ones and are the first to break +// if the tag dispatch runs out of positional slots. +class max_members +{ + public: + int m1{}, m2{}, m3{}, m4{}, m5{}, m6{}, m7{}, m8{}, m9{}, m10{}, m11{}, m12{}, m13{}, m14{}, m15{}, m16{}, m17{}, m18{}, m19{}, m20{}, m21{}, m22{}, m23{}, m24{}, m25{}, m26{}, m27{}, m28{}, m29{}, m30{}, m31{}, m32{}, m33{}, m34{}, m35{}, m36{}, m37{}, m38{}, m39{}, m40{}, m41{}, m42{}, m43{}, m44{}, m45{}, m46{}, m47{}, m48{}, m49{}, m50{}, m51{}, m52{}, m53{}, m54{}, m55{}, m56{}, m57{}, m58{}, m59{}, m60{}, m61{}, m62{}, m63{}; + + NLOHMANN_DEFINE_TYPE_INTRUSIVE(max_members, m1, m2, m3, m4, m5, m6, m7, m8, m9, m10, m11, m12, m13, m14, m15, m16, m17, m18, m19, m20, m21, m22, m23, m24, m25, m26, m27, m28, m29, m30, m31, m32, m33, m34, m35, m36, m37, m38, m39, m40, m41, m42, m43, m44, m45, m46, m47, m48, m49, m50, m51, m52, m53, m54, m55, m56, m57, m58, m59, m60, m61, m62, m63) +}; + +class max_members_base +{ + public: + int base_value = 0; + + NLOHMANN_DEFINE_TYPE_INTRUSIVE(max_members_base, base_value) +}; + +class max_members_derived : public max_members_base +{ + public: + int m1{}, m2{}, m3{}, m4{}, m5{}, m6{}, m7{}, m8{}, m9{}, m10{}, m11{}, m12{}, m13{}, m14{}, m15{}, m16{}, m17{}, m18{}, m19{}, m20{}, m21{}, m22{}, m23{}, m24{}, m25{}, m26{}, m27{}, m28{}, m29{}, m30{}, m31{}, m32{}, m33{}, m34{}, m35{}, m36{}, m37{}, m38{}, m39{}, m40{}, m41{}, m42{}, m43{}, m44{}, m45{}, m46{}, m47{}, m48{}, m49{}, m50{}, m51{}, m52{}, m53{}, m54{}, m55{}, m56{}, m57{}, m58{}, m59{}, m60{}, m61{}, m62{}, m63{}; + + NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE(max_members_derived, max_members_base, m1, m2, m3, m4, m5, m6, m7, m8, m9, m10, m11, m12, m13, m14, m15, m16, m17, m18, m19, m20, m21, m22, m23, m24, m25, m26, m27, m28, m29, m30, m31, m32, m33, m34, m35, m36, m37, m38, m39, m40, m41, m42, m43, m44, m45, m46, m47, m48, m49, m50, m51, m52, m53, m54, m55, m56, m57, m58, m59, m60, m61, m62, m63) +}; + +// User macros named like the dispatch suffixes (EMPTY is a common empty-macro +// idiom) must not leak into the NLOHMANN_DEFINE_TYPE_* dispatch. +#define EMPTY +#define MEMBERS clobbered_by_user_macro + +class dispatch_with_user_macros_empty +{ + public: + NLOHMANN_DEFINE_TYPE_INTRUSIVE(dispatch_with_user_macros_empty) +}; + +class dispatch_with_user_macros_members +{ + public: + int value = 0; + + NLOHMANN_DEFINE_TYPE_INTRUSIVE(dispatch_with_user_macros_members, value) +}; + +class dispatch_with_user_macros_derived_empty : public dispatch_with_user_macros_members +{ +}; +// NOLINTNEXTLINE(misc-use-internal-linkage) +NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE(dispatch_with_user_macros_derived_empty, dispatch_with_user_macros_members) + +class dispatch_with_user_macros_derived_members : public dispatch_with_user_macros_members +{ + public: + int own = 0; +}; +// NOLINTNEXTLINE(misc-use-internal-linkage) +NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE(dispatch_with_user_macros_derived_members, dispatch_with_user_macros_members, own) + +// testing for the macros also keeps -Wunused-macros from rejecting them +#if !defined(EMPTY) || !defined(MEMBERS) + #error "EMPTY and MEMBERS must stay defined for the tests above" +#endif +#undef EMPTY +#undef MEMBERS + } // namespace persons TEST_CASE_TEMPLATE("Serialization/deserialization via NLOHMANN_DEFINE_TYPE_INTRUSIVE and NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE", Pair, // NOLINT(readability-math-missing-parentheses, bugprone-throwing-static-initialization) @@ -1191,3 +1378,183 @@ TEST_CASE_TEMPLATE("Serialization of non-default-constructible classes via NLOHM } } } + +// Regression tests for issue #4041: NLOHMANN_DEFINE_TYPE_* and +// NLOHMANN_DEFINE_DERIVED_TYPE_* macros must compile and produce valid +// (empty, or base-only for the derived case) JSON objects when no member +// arguments are given, on every supported C++ standard. +TEST_CASE_TEMPLATE("Serialization/deserialization of zero-member types via NLOHMANN_DEFINE_TYPE_* (issue #4041)", Json, // NOLINT(readability-math-missing-parentheses, bugprone-throwing-static-initialization) + nlohmann::json, nlohmann::ordered_json) +{ + constexpr bool is_ordered = std::is_same::value; + const char* const derived_dump = is_ordered + ? R"({"age":1,"name":"Erik","metadata":null})" + : R"({"age":1,"metadata":null,"name":"Erik"})"; + + SECTION("NLOHMANN_DEFINE_TYPE_INTRUSIVE with zero members") + { + persons::empty_intrusive obj{}; + Json j = obj; + CHECK(j.dump() == "{}"); + CHECK(j.template get() == obj); + } + + SECTION("NLOHMANN_DEFINE_TYPE_INTRUSIVE_WITH_DEFAULT with zero members") + { + persons::empty_intrusive_with_default obj{}; + Json j = obj; + CHECK(j.dump() == "{}"); + CHECK(j.template get() == obj); + } + + SECTION("NLOHMANN_DEFINE_TYPE_INTRUSIVE_ONLY_SERIALIZE with zero members") + { + const persons::empty_intrusive_only_serialize obj{}; + Json j = obj; + CHECK(j.dump() == "{}"); + } + + SECTION("NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE with zero members") + { + persons::empty_non_intrusive obj{}; + Json j = obj; + CHECK(j.dump() == "{}"); + CHECK(j.template get() == obj); + } + + SECTION("NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE_WITH_DEFAULT with zero members") + { + persons::empty_non_intrusive_with_default obj{}; + Json j = obj; + CHECK(j.dump() == "{}"); + CHECK(j.template get() == obj); + } + + SECTION("NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE_ONLY_SERIALIZE with zero members") + { + const persons::empty_non_intrusive_only_serialize obj{}; + Json j = obj; + CHECK(j.dump() == "{}"); + } + + SECTION("NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE with zero own members") + { + persons::empty_derived_intrusive obj{"Erik", 1, nullptr}; + Json j = obj; + CHECK(j.dump() == derived_dump); + CHECK(j.template get() == obj); + } + + SECTION("NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE_WITH_DEFAULT with zero own members") + { + persons::empty_derived_intrusive_with_default obj{"Erik", 1, nullptr}; + Json j = obj; + CHECK(j.dump() == derived_dump); + CHECK(j.template get() == obj); + } + + SECTION("NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE_ONLY_SERIALIZE with zero own members") + { + const persons::empty_derived_intrusive_only_serialize obj{"Erik", 1, nullptr}; + Json j = obj; + CHECK(j.dump() == derived_dump); + } + + SECTION("NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE with zero own members") + { + persons::empty_derived_non_intrusive obj{"Erik", 1, nullptr}; + Json j = obj; + CHECK(j.dump() == derived_dump); + CHECK(j.template get() == obj); + } + + SECTION("NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_WITH_DEFAULT with zero own members") + { + persons::empty_derived_non_intrusive_with_default obj{"Erik", 1, nullptr}; + Json j = obj; + CHECK(j.dump() == derived_dump); + CHECK(j.template get() == obj); + } + + SECTION("NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_ONLY_SERIALIZE with zero own members") + { + const persons::empty_derived_non_intrusive_only_serialize obj{"Erik", 1, nullptr}; + Json j = obj; + CHECK(j.dump() == derived_dump); + } +} + +// Regression test for the argument-count dispatch added for issue #4041: the +// documented maximum of 63 members must keep working, including for the +// derived-type macros whose Type,BaseType prefix consumes two dispatch slots. +TEST_CASE_TEMPLATE("Serialization/deserialization of maximum-member-count types via NLOHMANN_DEFINE_TYPE_*", Json, // NOLINT(readability-math-missing-parentheses, bugprone-throwing-static-initialization) + nlohmann::json, nlohmann::ordered_json) +{ + SECTION("NLOHMANN_DEFINE_TYPE_INTRUSIVE with 63 members") + { + persons::max_members obj{}; + obj.m1 = 1; + obj.m63 = 63; + Json j = obj; + CHECK(j.size() == 63); + const auto obj2 = j.template get(); + CHECK(obj2.m1 == 1); + CHECK(obj2.m63 == 63); + } + + SECTION("NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE with 63 own members") + { + persons::max_members_derived obj{}; + obj.base_value = 7; + obj.m1 = 1; + obj.m63 = 63; + Json j = obj; + CHECK(j.size() == 64); + const auto obj2 = j.template get(); + CHECK(obj2.base_value == 7); + CHECK(obj2.m1 == 1); + CHECK(obj2.m63 == 63); + } +} + +TEST_CASE_TEMPLATE("NLOHMANN_DEFINE_TYPE_* dispatch is unaffected by user macros named EMPTY or MEMBERS", Json, // NOLINT(readability-math-missing-parentheses, bugprone-throwing-static-initialization) + nlohmann::json, nlohmann::ordered_json) +{ + SECTION("zero members") + { + const persons::dispatch_with_user_macros_empty obj{}; + const Json j = obj; + CHECK(j == Json::object()); + CHECK_NOTHROW(j.template get()); + } + + SECTION("one member") + { + persons::dispatch_with_user_macros_members obj{}; + obj.value = 42; + const Json j = obj; + CHECK(j == Json({{"value", 42}})); + CHECK(j.template get().value == 42); + } + + SECTION("derived with zero own members") + { + persons::dispatch_with_user_macros_derived_empty obj{}; + obj.value = 42; + const Json j = obj; + CHECK(j == Json({{"value", 42}})); + CHECK(j.template get().value == 42); + } + + SECTION("derived with own members") + { + persons::dispatch_with_user_macros_derived_members obj{}; + obj.value = 42; + obj.own = 7; + const Json j = obj; + CHECK(j == Json({{"value", 42}, {"own", 7}})); + const auto obj2 = j.template get(); + CHECK(obj2.value == 42); + CHECK(obj2.own == 7); + } +} From d19f7f5dce601afc81cc1987ee124d4a05522c73 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 25 Sep 2026 20:45:28 +0200 Subject: [PATCH 50/64] Fix BSON conformance issue (#5185) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * :bug: fix BSON conformance issue Signed-off-by: Niels Lohmann * :bug: fix BSON conformance issue Signed-off-by: Niels Lohmann * :bug: reject ill-formed UTF-8 in CBOR/MessagePack/BSON text strings at decode time (#5531) from_cbor()/from_msgpack()/from_bson() copied the raw bytes of a decoded text string into the resulting json value without any UTF-8 validation, even though RFC 8949 §3.1 (CBOR) and the MessagePack/BSON specifications all require text strings to be valid UTF-8. Malformed input only failed later, if the value was dump()'d, with a type_error.316 - so the allow_exceptions=false pattern used specifically to get a discarded sentinel instead of an exception did not discard this category of malformed input, unlike every other kind of malformed binary input this library rejects at decode time (see #5529). Fix this at the single choke point shared by BSON/CBOR/MessagePack/UBJSON string reads, binary_reader::get_string(): validate the bytes with the UTF-8 DFA right after they are read, and report failures the same way as every other binary_reader error (parse_error.113), so allow_exceptions and strict discarding behave consistently. get_binary()/binary blob reads are untouched and still accept arbitrary bytes, since only text strings are required to be UTF-8. There were two independent implementations of a UTF-8 validator: the lexer's streaming scanner, and the serializer's Hoehrmann DFA used by dump_escaped_impl(). Rather than write a third, the serializer's decode() function, its utf8d table and the UTF8_ACCEPT/UTF8_REJECT constants are extracted into detail/string_utils.hpp (a low-level header already included before both detail/input/ and detail/output/), alongside a new is_valid_utf8() helper built on the same decode() step. serializer.hpp's dump_escaped_impl() now calls the shared decode(), so there is exactly one UTF-8 validator in the codebase; dump()'s exact type_error.316 messages and byte-index reporting are unchanged (see the added regression-guard test in unit-serialization.cpp). Claude-Session: https://claude.ai/code/session_01N4RQ1Ahan5YAGbnAQGjZTY Signed-off-by: Niels Lohmann Co-authored-by: Claude Sonnet 5 * :zap: validate only newly read bytes of binary-format strings get_string() validated the whole result after each call, but get_bytes() appends to it and CBOR indefinite-length strings collect all chunks in the same result, so every chunk re-validated everything read before it. An input of many small chunks took quadratic time (80000 one-byte chunks, 160 KB of input, took about 7 seconds). Only the newly read bytes are validated now, which also matches RFC 8949's requirement that every chunk is valid UTF-8 on its own. Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann Co-authored-by: Claude Sonnet 5 --- .../docs/features/binary_formats/bson.md | 9 + .../docs/features/binary_formats/cbor.md | 10 + .../features/binary_formats/messagepack.md | 8 + docs/mkdocs/docs/home/exceptions.md | 6 +- .../nlohmann/detail/input/binary_reader.hpp | 42 +++- include/nlohmann/detail/output/serializer.hpp | 59 +---- include/nlohmann/detail/string_utils.hpp | 100 +++++++++ single_include/nlohmann/json.hpp | 204 ++++++++++++------ tests/src/unit-bson.cpp | 73 +++++++ tests/src/unit-cbor.cpp | 53 +++++ tests/src/unit-msgpack.cpp | 21 ++ tests/src/unit-serialization.cpp | 10 + 12 files changed, 470 insertions(+), 125 deletions(-) diff --git a/docs/mkdocs/docs/features/binary_formats/bson.md b/docs/mkdocs/docs/features/binary_formats/bson.md index 95c82e873..6f5603c8c 100644 --- a/docs/mkdocs/docs/features/binary_formats/bson.md +++ b/docs/mkdocs/docs/features/binary_formats/bson.md @@ -109,6 +109,15 @@ The library maps BSON record types to JSON value types as follows: If BSON input must be validated for strict specification compliance, validate it separately before passing it to `from_bson()`. +!!! warning "UTF-8 validation of string values" + + The BSON specification requires `string` values (type `0x02`) to be valid UTF-8. This library validates the + bytes of every such string at decode time and rejects ill-formed UTF-8 with a + [`parse_error.113`](../../home/exceptions.md#jsonexceptionparse_error113) exception (or, with `allow_exceptions` + set to `false`, a discarded value), rather than only failing later when the resulting value is dumped. Element + (key) names and `binary` values (type `0x05`) are unaffected and are never validated, since they are read + byte-by-byte as a C string, or are not required to hold text, respectively. + ??? example ```cpp diff --git a/docs/mkdocs/docs/features/binary_formats/cbor.md b/docs/mkdocs/docs/features/binary_formats/cbor.md index 670a23455..e4c257e27 100644 --- a/docs/mkdocs/docs/features/binary_formats/cbor.md +++ b/docs/mkdocs/docs/features/binary_formats/cbor.md @@ -176,6 +176,16 @@ The library maps CBOR types to JSON value types as follows: CBOR allows map keys of any type, whereas JSON only allows strings as keys in object values. Therefore, CBOR maps with keys other than UTF-8 strings are rejected. +!!! warning "UTF-8 validation of text strings" + + [RFC 8949, Section 3.1](https://www.rfc-editor.org/rfc/rfc8949.html#section-3.1) requires CBOR text strings + (major type 3) to be valid UTF-8. This library validates the bytes of every text string (object keys included) at + decode time and rejects ill-formed UTF-8 with a + [`parse_error.113`](../../home/exceptions.md#jsonexceptionparse_error113) exception (or, with + `allow_exceptions` set to `false`, a discarded value), rather than only failing later when the resulting value is + dumped. Byte strings (major type 2) are unaffected and are never validated, since they are not required to hold + text. + !!! warning "Tagged items" Tagged items (0xC0..0xDB) will throw a parse error by default. They can be ignored by passing `cbor_tag_handler_t::ignore` to function `from_cbor`, in which case the tag is skipped and the enclosed data item is parsed on its own. They can be stored by passing `cbor_tag_handler_t::store` to function `from_cbor`. Note that no tag is ever interpreted: for instance, a text string tagged with tag 0 (date/time) stays a string. diff --git a/docs/mkdocs/docs/features/binary_formats/messagepack.md b/docs/mkdocs/docs/features/binary_formats/messagepack.md index bd0c840f2..a434909c4 100644 --- a/docs/mkdocs/docs/features/binary_formats/messagepack.md +++ b/docs/mkdocs/docs/features/binary_formats/messagepack.md @@ -136,6 +136,14 @@ The library maps MessagePack types to JSON value types as follows: Any MessagePack output created by `to_msgpack` can be successfully parsed by `from_msgpack`. +!!! warning "UTF-8 validation of string values" + + The MessagePack specification requires `str` values (`fixstr`, `str 8`, `str 16`, `str 32`) to be valid UTF-8. + This library validates the bytes of every such string (object keys included) at decode time and rejects + ill-formed UTF-8 with a [`parse_error.113`](../../home/exceptions.md#jsonexceptionparse_error113) exception (or, + with `allow_exceptions` set to `false`, a discarded value), rather than only failing later when the resulting + value is dumped. `bin`/`ext`/`fixext` values are unaffected and are never validated, since they are not required + to hold text. ??? example diff --git a/docs/mkdocs/docs/home/exceptions.md b/docs/mkdocs/docs/home/exceptions.md index 9a7698b2f..ee76596f6 100644 --- a/docs/mkdocs/docs/home/exceptions.md +++ b/docs/mkdocs/docs/home/exceptions.md @@ -340,7 +340,8 @@ An unexpected byte was read in a [binary format](../features/binary_formats/inde ### json.exception.parse_error.113 A string could not be read from a [binary format](../features/binary_formats/index.md): either a value that is not a -string was read where one was required (for instance as a map key), or the string's length specification is invalid. +string was read where one was required (for instance as a map key), the string's length specification is invalid, or +the string's bytes are not valid UTF-8. !!! failure "Example messages" @@ -356,6 +357,9 @@ string was read where one was required (for instance as a map key), or the strin ``` [json.exception.parse_error.113] parse error at byte 3: syntax error while parsing BJData string: string length must not be negative ``` + ``` + [json.exception.parse_error.113] parse error at byte 3: syntax error while parsing CBOR string: invalid string: ill-formed UTF-8 byte + ``` ### json.exception.parse_error.114 diff --git a/include/nlohmann/detail/input/binary_reader.hpp b/include/nlohmann/detail/input/binary_reader.hpp index df46eea58..4132c03ca 100644 --- a/include/nlohmann/detail/input/binary_reader.hpp +++ b/include/nlohmann/detail/input/binary_reader.hpp @@ -32,6 +32,7 @@ #include #include #include +#include #include NLOHMANN_JSON_NAMESPACE_BEGIN @@ -432,7 +433,21 @@ class binary_reader exception_message(input_format_t::bson, concat("string length must be at least 1, is ", std::to_string(len)), "string"), nullptr)); } - return get_string(input_format_t::bson, len - static_cast(1), result) && get() != char_traits::eof(); + if (JSON_HEDLEY_UNLIKELY(!get_string(input_format_t::bson, len - static_cast(1), result))) + { + return false; + } + + if (JSON_HEDLEY_UNLIKELY(get() != 0x00)) + { + auto last_token = get_token_string(); + return sax->parse_error(chars_read, last_token, parse_error::create(112, chars_read, + exception_message(input_format_t::bson, + "BSON string is not null-terminated", + "string"), nullptr)); + } + + return true; } /*! @@ -550,8 +565,6 @@ class binary_reader } } - - ////////// // CBOR // ////////// @@ -3304,7 +3317,28 @@ class binary_reader const NumberType len, string_t& result) { - return get_bytes(format, len, "string", result); + // get_bytes() appends to result, and CBOR indefinite-length strings + // collect all their chunks in the same result; validating only the + // newly read bytes keeps the check linear in the input size + const std::size_t old_size = result.size(); + if (JSON_HEDLEY_UNLIKELY(!get_bytes(format, len, "string", result))) + { + return false; + } + + // RFC 8949 (CBOR) §3.1 and the MessagePack/BSON/UBJSON specifications + // all require text strings to be valid UTF-8; reject anything else + // right here so malformed input is caught at decode time instead of + // only surfacing later as a type_error.316 when the value is dumped + // (which would defeat allow_exceptions=false / strict discarding). + if (JSON_HEDLEY_UNLIKELY(!is_valid_utf8(result, old_size))) + { + return sax->parse_error(chars_read, get_token_string(), + parse_error::create(113, chars_read, + exception_message(format, "invalid string: ill-formed UTF-8 byte", "string"), nullptr)); + } + + return true; } /*! diff --git a/include/nlohmann/detail/output/serializer.hpp b/include/nlohmann/detail/output/serializer.hpp index 7c38276ce..f968b6001 100644 --- a/include/nlohmann/detail/output/serializer.hpp +++ b/include/nlohmann/detail/output/serializer.hpp @@ -32,6 +32,7 @@ #include #include #include +#include #include NLOHMANN_JSON_NAMESPACE_BEGIN @@ -58,8 +59,6 @@ class serializer using number_integer_t = typename BasicJsonType::number_integer_t; using number_unsigned_t = typename BasicJsonType::number_unsigned_t; using binary_char_t = typename BasicJsonType::binary_t::value_type; - static constexpr std::uint8_t UTF8_ACCEPT = 0; - static constexpr std::uint8_t UTF8_REJECT = 1; public: /*! @@ -1592,62 +1591,6 @@ class serializer } } - /*! - @brief check whether a string is UTF-8 encoded - - The function checks each byte of a string whether it is UTF-8 encoded. The - result of the check is stored in the @a state parameter. The function must - be called initially with state 0 (accept). State 1 means the string must - be rejected, because the current byte is not allowed. If the string is - completely processed, but the state is non-zero, the string ended - prematurely; that is, the last byte indicated more bytes should have - followed. - - @param[in,out] state the state of the decoding - @param[in,out] codep codepoint (valid only if resulting state is UTF8_ACCEPT) - @param[in] byte next byte to decode - @return new state - - @note The function has been edited: a std::array is used. - - @copyright Copyright (c) 2008-2009 Bjoern Hoehrmann - @sa http://bjoern.hoehrmann.de/utf-8/decoder/dfa/ - */ - static std::uint8_t decode(std::uint8_t& state, std::uint32_t& codep, const std::uint8_t byte) noexcept - { - static const std::array utf8d = - { - { - 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, // 00..1F - 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, // 20..3F - 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, // 40..5F - 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, // 60..7F - 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, // 80..9F - 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, // A0..BF - 8, 8, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, // C0..DF - 0xA, 0x3, 0x3, 0x3, 0x3, 0x3, 0x3, 0x3, 0x3, 0x3, 0x3, 0x3, 0x3, 0x4, 0x3, 0x3, // E0..EF - 0xB, 0x6, 0x6, 0x6, 0x5, 0x8, 0x8, 0x8, 0x8, 0x8, 0x8, 0x8, 0x8, 0x8, 0x8, 0x8, // F0..FF - 0x0, 0x1, 0x2, 0x3, 0x5, 0x8, 0x7, 0x1, 0x1, 0x1, 0x4, 0x6, 0x1, 0x1, 0x1, 0x1, // s0..s0 - 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 0, 1, 1, 1, 1, 1, 0, 1, 0, 1, 1, 1, 1, 1, 1, // s1..s2 - 1, 2, 1, 1, 1, 1, 1, 2, 1, 2, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 2, 1, 1, 1, 1, 1, 1, 1, 1, // s3..s4 - 1, 2, 1, 1, 1, 1, 1, 1, 1, 2, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 3, 1, 3, 1, 1, 1, 1, 1, 1, // s5..s6 - 1, 3, 1, 1, 1, 1, 1, 3, 1, 3, 1, 1, 1, 1, 1, 1, 1, 3, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1 // s7..s8 - } - }; - - JSON_ASSERT(static_cast(byte) < utf8d.size()); - const std::uint8_t type = utf8d[byte]; - - codep = (state != UTF8_ACCEPT) - ? (byte & 0x3fu) | (codep << 6u) - : (0xFFu >> type) & (byte); - - const std::size_t index = 256u + (static_cast(state) * 16u) + static_cast(type); - JSON_ASSERT(index < utf8d.size()); - state = utf8d[index]; - return state; - } - /* * Overload to make the compiler happy while it is instantiating * dump_integer for number_unsigned_t. diff --git a/include/nlohmann/detail/string_utils.hpp b/include/nlohmann/detail/string_utils.hpp index fe2f9109d..142943cd6 100644 --- a/include/nlohmann/detail/string_utils.hpp +++ b/include/nlohmann/detail/string_utils.hpp @@ -8,10 +8,13 @@ #pragma once +#include // array #include // size_t +#include // uint8_t, uint32_t #include // string, to_string #include +#include NLOHMANN_JSON_NAMESPACE_BEGIN namespace detail @@ -33,5 +36,102 @@ StringType to_string(std::size_t value) return result; } +/////////////////// +// UTF-8 decoding // +/////////////////// + +// UTF-8 decoder states used by decode() below +static constexpr std::uint8_t UTF8_ACCEPT = 0; +static constexpr std::uint8_t UTF8_REJECT = 1; + +/*! +@brief process a byte of a UTF-8 sequence + +This is a single-byte step of a "shift-based" UTF-8 decoder originally +written by Björn Hoehrmann. See +http://bjoern.hoehrmann.de/utf-8/decoder/dfa/ for details. + +This decoder is the single source of truth for UTF-8 validation in this +library: it is used both by the serializer (to escape and, in strict mode, +reject ill-formed UTF-8 when dumping a string) and by the binary readers +(to reject ill-formed UTF-8 in CBOR/MessagePack/BSON/UBJSON text strings at +decode time; see @ref is_valid_utf8 below). + +@param[in,out] state the current decoder state +@param[in,out] codep codepoint (valid only if resulting state is UTF8_ACCEPT) +@param[in] byte next byte to decode +@return new state + +@note Original source: http://bjoern.hoehrmann.de/utf-8/decoder/dfa/ +@sa http://bjoern.hoehrmann.de/utf-8/decoder/dfa/ +*/ +inline std::uint8_t decode(std::uint8_t& state, std::uint32_t& codep, const std::uint8_t byte) noexcept +{ + static const std::array utf8d = + { + { + 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, // 00..1F + 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, // 20..3F + 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, // 40..5F + 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, // 60..7F + 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, // 80..9F + 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, // A0..BF + 8, 8, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, // C0..DF + 0xA, 0x3, 0x3, 0x3, 0x3, 0x3, 0x3, 0x3, 0x3, 0x3, 0x3, 0x3, 0x3, 0x4, 0x3, 0x3, // E0..EF + 0xB, 0x6, 0x6, 0x6, 0x5, 0x8, 0x8, 0x8, 0x8, 0x8, 0x8, 0x8, 0x8, 0x8, 0x8, 0x8, // F0..FF + 0x0, 0x1, 0x2, 0x3, 0x5, 0x8, 0x7, 0x1, 0x1, 0x1, 0x4, 0x6, 0x1, 0x1, 0x1, 0x1, // s0..s0 + 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 0, 1, 1, 1, 1, 1, 0, 1, 0, 1, 1, 1, 1, 1, 1, // s1..s2 + 1, 2, 1, 1, 1, 1, 1, 2, 1, 2, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 2, 1, 1, 1, 1, 1, 1, 1, 1, // s3..s4 + 1, 2, 1, 1, 1, 1, 1, 1, 1, 2, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 3, 1, 3, 1, 1, 1, 1, 1, 1, // s5..s6 + 1, 3, 1, 1, 1, 1, 1, 3, 1, 3, 1, 1, 1, 1, 1, 1, 1, 3, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1 // s7..s8 + } + }; + + JSON_ASSERT(static_cast(byte) < utf8d.size()); + const std::uint8_t type = utf8d[byte]; + + codep = (state != UTF8_ACCEPT) + ? (byte & 0x3fu) | (codep << 6u) + : (0xFFu >> type) & (byte); + + const std::size_t index = 256u + (static_cast(state) * 16u) + static_cast(type); + JSON_ASSERT(index < utf8d.size()); + state = utf8d[index]; + return state; +} + +/*! +@brief check whether a string consists solely of valid UTF-8 + +Used by the CBOR/MessagePack/BSON/UBJSON binary readers to reject text +strings that are not valid UTF-8 at decode time (RFC 8949 §3.1 and the +MessagePack/BSON specifications all require text strings to be UTF-8), so +that malformed input is caught immediately instead of only surfacing later +as a type_error.316 when the resulting value is dumped. + +@param[in] s the string to check +@param[in] first index of the first byte to check; the bytes before it are + assumed to have been validated already and to end on a + code point boundary +@return whether @a s (from index @a first on) is valid UTF-8 +*/ +template +inline bool is_valid_utf8(const StringType& s, const std::size_t first = 0) noexcept +{ + std::uint8_t state = UTF8_ACCEPT; + std::uint32_t codepoint = 0; + + for (std::size_t i = first; i < s.size(); ++i) + { + decode(state, codepoint, static_cast(s[i])); + if (state == UTF8_REJECT) + { + return false; + } + } + + return state == UTF8_ACCEPT; +} + } // namespace detail NLOHMANN_JSON_NAMESPACE_END diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index e6fe169ef..647312562 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -6197,11 +6197,15 @@ NLOHMANN_JSON_NAMESPACE_END +#include // array #include // size_t +#include // uint8_t, uint32_t #include // string, to_string // #include +// #include + NLOHMANN_JSON_NAMESPACE_BEGIN namespace detail @@ -6223,6 +6227,103 @@ StringType to_string(std::size_t value) return result; } +/////////////////// +// UTF-8 decoding // +/////////////////// + +// UTF-8 decoder states used by decode() below +static constexpr std::uint8_t UTF8_ACCEPT = 0; +static constexpr std::uint8_t UTF8_REJECT = 1; + +/*! +@brief process a byte of a UTF-8 sequence + +This is a single-byte step of a "shift-based" UTF-8 decoder originally +written by Björn Hoehrmann. See +http://bjoern.hoehrmann.de/utf-8/decoder/dfa/ for details. + +This decoder is the single source of truth for UTF-8 validation in this +library: it is used both by the serializer (to escape and, in strict mode, +reject ill-formed UTF-8 when dumping a string) and by the binary readers +(to reject ill-formed UTF-8 in CBOR/MessagePack/BSON/UBJSON text strings at +decode time; see @ref is_valid_utf8 below). + +@param[in,out] state the current decoder state +@param[in,out] codep codepoint (valid only if resulting state is UTF8_ACCEPT) +@param[in] byte next byte to decode +@return new state + +@note Original source: http://bjoern.hoehrmann.de/utf-8/decoder/dfa/ +@sa http://bjoern.hoehrmann.de/utf-8/decoder/dfa/ +*/ +inline std::uint8_t decode(std::uint8_t& state, std::uint32_t& codep, const std::uint8_t byte) noexcept +{ + static const std::array utf8d = + { + { + 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, // 00..1F + 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, // 20..3F + 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, // 40..5F + 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, // 60..7F + 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, // 80..9F + 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, // A0..BF + 8, 8, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, // C0..DF + 0xA, 0x3, 0x3, 0x3, 0x3, 0x3, 0x3, 0x3, 0x3, 0x3, 0x3, 0x3, 0x3, 0x4, 0x3, 0x3, // E0..EF + 0xB, 0x6, 0x6, 0x6, 0x5, 0x8, 0x8, 0x8, 0x8, 0x8, 0x8, 0x8, 0x8, 0x8, 0x8, 0x8, // F0..FF + 0x0, 0x1, 0x2, 0x3, 0x5, 0x8, 0x7, 0x1, 0x1, 0x1, 0x4, 0x6, 0x1, 0x1, 0x1, 0x1, // s0..s0 + 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 0, 1, 1, 1, 1, 1, 0, 1, 0, 1, 1, 1, 1, 1, 1, // s1..s2 + 1, 2, 1, 1, 1, 1, 1, 2, 1, 2, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 2, 1, 1, 1, 1, 1, 1, 1, 1, // s3..s4 + 1, 2, 1, 1, 1, 1, 1, 1, 1, 2, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 3, 1, 3, 1, 1, 1, 1, 1, 1, // s5..s6 + 1, 3, 1, 1, 1, 1, 1, 3, 1, 3, 1, 1, 1, 1, 1, 1, 1, 3, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1 // s7..s8 + } + }; + + JSON_ASSERT(static_cast(byte) < utf8d.size()); + const std::uint8_t type = utf8d[byte]; + + codep = (state != UTF8_ACCEPT) + ? (byte & 0x3fu) | (codep << 6u) + : (0xFFu >> type) & (byte); + + const std::size_t index = 256u + (static_cast(state) * 16u) + static_cast(type); + JSON_ASSERT(index < utf8d.size()); + state = utf8d[index]; + return state; +} + +/*! +@brief check whether a string consists solely of valid UTF-8 + +Used by the CBOR/MessagePack/BSON/UBJSON binary readers to reject text +strings that are not valid UTF-8 at decode time (RFC 8949 §3.1 and the +MessagePack/BSON specifications all require text strings to be UTF-8), so +that malformed input is caught immediately instead of only surfacing later +as a type_error.316 when the resulting value is dumped. + +@param[in] s the string to check +@param[in] first index of the first byte to check; the bytes before it are + assumed to have been validated already and to end on a + code point boundary +@return whether @a s (from index @a first on) is valid UTF-8 +*/ +template +inline bool is_valid_utf8(const StringType& s, const std::size_t first = 0) noexcept +{ + std::uint8_t state = UTF8_ACCEPT; + std::uint32_t codepoint = 0; + + for (std::size_t i = first; i < s.size(); ++i) + { + decode(state, codepoint, static_cast(s[i])); + if (state == UTF8_REJECT) + { + return false; + } + } + + return state == UTF8_ACCEPT; +} + } // namespace detail NLOHMANN_JSON_NAMESPACE_END @@ -12619,6 +12720,8 @@ NLOHMANN_JSON_NAMESPACE_END // #include +// #include + // #include @@ -13020,7 +13123,21 @@ class binary_reader exception_message(input_format_t::bson, concat("string length must be at least 1, is ", std::to_string(len)), "string"), nullptr)); } - return get_string(input_format_t::bson, len - static_cast(1), result) && get() != char_traits::eof(); + if (JSON_HEDLEY_UNLIKELY(!get_string(input_format_t::bson, len - static_cast(1), result))) + { + return false; + } + + if (JSON_HEDLEY_UNLIKELY(get() != 0x00)) + { + auto last_token = get_token_string(); + return sax->parse_error(chars_read, last_token, parse_error::create(112, chars_read, + exception_message(input_format_t::bson, + "BSON string is not null-terminated", + "string"), nullptr)); + } + + return true; } /*! @@ -13138,8 +13255,6 @@ class binary_reader } } - - ////////// // CBOR // ////////// @@ -15892,7 +16007,28 @@ class binary_reader const NumberType len, string_t& result) { - return get_bytes(format, len, "string", result); + // get_bytes() appends to result, and CBOR indefinite-length strings + // collect all their chunks in the same result; validating only the + // newly read bytes keeps the check linear in the input size + const std::size_t old_size = result.size(); + if (JSON_HEDLEY_UNLIKELY(!get_bytes(format, len, "string", result))) + { + return false; + } + + // RFC 8949 (CBOR) §3.1 and the MessagePack/BSON/UBJSON specifications + // all require text strings to be valid UTF-8; reject anything else + // right here so malformed input is caught at decode time instead of + // only surfacing later as a type_error.316 when the value is dumped + // (which would defeat allow_exceptions=false / strict discarding). + if (JSON_HEDLEY_UNLIKELY(!is_valid_utf8(result, old_size))) + { + return sax->parse_error(chars_read, get_token_string(), + parse_error::create(113, chars_read, + exception_message(format, "invalid string: ill-formed UTF-8 byte", "string"), nullptr)); + } + + return true; } /*! @@ -22726,6 +22862,8 @@ NLOHMANN_JSON_NAMESPACE_END // #include +// #include + // #include @@ -22753,8 +22891,6 @@ class serializer using number_integer_t = typename BasicJsonType::number_integer_t; using number_unsigned_t = typename BasicJsonType::number_unsigned_t; using binary_char_t = typename BasicJsonType::binary_t::value_type; - static constexpr std::uint8_t UTF8_ACCEPT = 0; - static constexpr std::uint8_t UTF8_REJECT = 1; public: /*! @@ -24287,62 +24423,6 @@ class serializer } } - /*! - @brief check whether a string is UTF-8 encoded - - The function checks each byte of a string whether it is UTF-8 encoded. The - result of the check is stored in the @a state parameter. The function must - be called initially with state 0 (accept). State 1 means the string must - be rejected, because the current byte is not allowed. If the string is - completely processed, but the state is non-zero, the string ended - prematurely; that is, the last byte indicated more bytes should have - followed. - - @param[in,out] state the state of the decoding - @param[in,out] codep codepoint (valid only if resulting state is UTF8_ACCEPT) - @param[in] byte next byte to decode - @return new state - - @note The function has been edited: a std::array is used. - - @copyright Copyright (c) 2008-2009 Bjoern Hoehrmann - @sa http://bjoern.hoehrmann.de/utf-8/decoder/dfa/ - */ - static std::uint8_t decode(std::uint8_t& state, std::uint32_t& codep, const std::uint8_t byte) noexcept - { - static const std::array utf8d = - { - { - 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, // 00..1F - 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, // 20..3F - 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, // 40..5F - 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, // 60..7F - 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, 9, // 80..9F - 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, // A0..BF - 8, 8, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, // C0..DF - 0xA, 0x3, 0x3, 0x3, 0x3, 0x3, 0x3, 0x3, 0x3, 0x3, 0x3, 0x3, 0x3, 0x4, 0x3, 0x3, // E0..EF - 0xB, 0x6, 0x6, 0x6, 0x5, 0x8, 0x8, 0x8, 0x8, 0x8, 0x8, 0x8, 0x8, 0x8, 0x8, 0x8, // F0..FF - 0x0, 0x1, 0x2, 0x3, 0x5, 0x8, 0x7, 0x1, 0x1, 0x1, 0x4, 0x6, 0x1, 0x1, 0x1, 0x1, // s0..s0 - 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 0, 1, 1, 1, 1, 1, 0, 1, 0, 1, 1, 1, 1, 1, 1, // s1..s2 - 1, 2, 1, 1, 1, 1, 1, 2, 1, 2, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 2, 1, 1, 1, 1, 1, 1, 1, 1, // s3..s4 - 1, 2, 1, 1, 1, 1, 1, 1, 1, 2, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 3, 1, 3, 1, 1, 1, 1, 1, 1, // s5..s6 - 1, 3, 1, 1, 1, 1, 1, 3, 1, 3, 1, 1, 1, 1, 1, 1, 1, 3, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1 // s7..s8 - } - }; - - JSON_ASSERT(static_cast(byte) < utf8d.size()); - const std::uint8_t type = utf8d[byte]; - - codep = (state != UTF8_ACCEPT) - ? (byte & 0x3fu) | (codep << 6u) - : (0xFFu >> type) & (byte); - - const std::size_t index = 256u + (static_cast(state) * 16u) + static_cast(type); - JSON_ASSERT(index < utf8d.size()); - state = utf8d[index]; - return state; - } - /* * Overload to make the compiler happy while it is instantiating * dump_integer for number_unsigned_t. diff --git a/tests/src/unit-bson.cpp b/tests/src/unit-bson.cpp index 669a4bfe1..4808c3166 100644 --- a/tests/src/unit-bson.cpp +++ b/tests/src/unit-bson.cpp @@ -1697,3 +1697,76 @@ TEST_CASE("BSON roundtrips" * doctest::skip()) } } } + +TEST_CASE("Invalid document size handling") +{ + SECTION("document size must be at least 5") + { + std::vector const v = {0x04, 0x00, 0x00, 0x00, 0x00}; + json _; + CHECK_THROWS_WITH_AS(_ = json::from_bson(v), "[json.exception.parse_error.112] parse error at byte 5: syntax error while parsing BSON document: document size 4 does not match the number of bytes read (5)", json::parse_error&); + CHECK(json::from_bson(v, true, false).is_discarded()); + } + + SECTION("declared document size must match consumed bytes (extra trailing element)") + { + // Declares 5-byte empty document but appends an int32 element after the declared end. + std::vector const v = + { + 0x05, 0x00, 0x00, 0x00, + 0x10, 'a', 'd', 'm', 'i', 'n', 0x00, + 0x01, 0x00, 0x00, 0x00, + 0x00 + }; + json _; + CHECK_THROWS_WITH_AS(_ = json::from_bson(v), "[json.exception.parse_error.112] parse error at byte 16: syntax error while parsing BSON document: document size 5 does not match the number of bytes read (16)", json::parse_error&); + CHECK(json::from_bson(v, true, false).is_discarded()); + } + + SECTION("declared document size must match consumed bytes (premature terminator)") + { + // Declares 32-byte document but only contains the size field followed by an immediate terminator. + std::vector const v = + { + 0x20, 0x00, 0x00, 0x00, + 0x00 + }; + json _; + CHECK_THROWS_WITH_AS(_ = json::from_bson(v), "[json.exception.parse_error.112] parse error at byte 5: syntax error while parsing BSON document: document size 32 does not match the number of bytes read (5)", json::parse_error&); + CHECK(json::from_bson(v, true, false).is_discarded()); + } + + SECTION("array declared size must match consumed bytes") + { + // Outer object contains an array "a" that declares 5 bytes (empty) but + // actually contains an int32 element before its terminator. + std::vector const v = + { + 0x14, 0x00, 0x00, 0x00, // object size = 20 + 0x04, 'a', 0x00, // key "a", array type + 0x05, 0x00, 0x00, 0x00, // array declared size = 5 (empty) + 0x10, '0', 0x00, 0x01, 0x00, 0x00, 0x00, // extra int32 element "0" = 1 + 0x00, // array terminator + 0x00 // object terminator + }; + json _; + CHECK_THROWS_WITH_AS(_ = json::from_bson(v), "[json.exception.parse_error.112] parse error at byte 19: syntax error while parsing BSON document: document size 5 does not match the number of bytes read (12)", json::parse_error&); + CHECK(json::from_bson(v, true, false).is_discarded()); + } + + SECTION("BSON string must end with 0x00") + { + // Length-prefixed string whose terminator byte is 'X' (0x58), not 0x00. + std::vector const v = + { + 0x0F, 0x00, 0x00, 0x00, + 0x02, 's', 0x00, + 0x02, 0x00, 0x00, 0x00, + 'A', 'X', + 0x00 + }; + json _; + CHECK_THROWS_WITH_AS(_ = json::from_bson(v), "[json.exception.parse_error.112] parse error at byte 13: syntax error while parsing BSON string: BSON string is not null-terminated", json::parse_error&); + CHECK(json::from_bson(v, true, false).is_discarded()); + } +} diff --git a/tests/src/unit-cbor.cpp b/tests/src/unit-cbor.cpp index 4c9107517..9c799b0eb 100644 --- a/tests/src/unit-cbor.cpp +++ b/tests/src/unit-cbor.cpp @@ -1833,6 +1833,59 @@ TEST_CASE("CBOR") CHECK(json::from_cbor(std::vector({0xa1, 0xff, 0x01}), true, false).is_discarded()); } + SECTION("invalid UTF-8 in string (see #5529)") + { + // a two-character text string (major type 3) whose bytes are not + // valid UTF-8 (0xC0 0xAE is an overlong encoding of '.') must be + // rejected at decode time, matching every other kind of + // malformed binary input, rather than only failing later when + // the resulting value is dumped + json _; + CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector({0x62, 0xc0, 0xae})), "[json.exception.parse_error.113] parse error at byte 3: syntax error while parsing CBOR string: invalid string: ill-formed UTF-8 byte", json::parse_error&); + CHECK(json::from_cbor(std::vector({0x62, 0xc0, 0xae}), true, false).is_discarded()); + + // a CBOR byte string (major type 2) with the very same bytes is + // NOT text and must still be accepted as-is + CHECK_NOTHROW(_ = json::from_cbor(std::vector({0x42, 0xc0, 0xae}))); + CHECK(_ == json::binary(std::vector({0xc0, 0xae}))); + + // valid UTF-8 must still round-trip + const json j = "h\xc3\xa9llo, w\xc3\xb6rld! \xe6\x97\xa5\xe6\x9c\xac\xe8\xaa\x9e"; // héllo, wörld! 日本語 + CHECK(json::from_cbor(json::to_cbor(j)) == j); + } + + SECTION("invalid UTF-8 in indefinite-length string") + { + json _; + + // every chunk must be valid UTF-8 on its own (RFC 8949, Section + // 3.2.3), so a code point split across two chunks is rejected + CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector({0x7f, 0x61, 0xc3, 0x61, 0xa9, 0xff})), "[json.exception.parse_error.113] parse error at byte 3: syntax error while parsing CBOR string: invalid string: ill-formed UTF-8 byte", json::parse_error&); + CHECK(json::from_cbor(std::vector({0x7f, 0x61, 0xc3, 0x61, 0xa9, 0xff}), true, false).is_discarded()); + + // an ill-formed later chunk is rejected after valid ones + CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector({0x7f, 0x62, 0xc3, 0xa9, 0x62, 0xc0, 0xae, 0xff})), "[json.exception.parse_error.113] parse error at byte 7: syntax error while parsing CBOR string: invalid string: ill-formed UTF-8 byte", json::parse_error&); + + // valid multi-byte chunks are accepted + CHECK(json::from_cbor(std::vector({0x7f, 0x62, 0xc3, 0xa9, 0x62, 0xc3, 0xb6, 0xff})) == "\xc3\xa9\xc3\xb6"); + } + + SECTION("many chunks in indefinite-length string") + { + // only the newly read chunk is validated, not the whole string + // collected so far; validating the latter made this input take + // quadratic time (about ten seconds for 100000 chunks) + constexpr std::size_t chunks = 100000; + std::vector v{0x7f}; + for (std::size_t i = 0; i < chunks; ++i) + { + v.push_back(0x61); + v.push_back('a'); + } + v.push_back(0xff); + CHECK(json::from_cbor(v) == std::string(chunks, 'a')); + } + SECTION("strict mode") { std::vector const vec = {0xf6, 0xf6}; diff --git a/tests/src/unit-msgpack.cpp b/tests/src/unit-msgpack.cpp index a8892081d..805b03a13 100644 --- a/tests/src/unit-msgpack.cpp +++ b/tests/src/unit-msgpack.cpp @@ -1554,6 +1554,27 @@ TEST_CASE("MessagePack") CHECK(json::from_msgpack(std::vector({0x81, 0xff, 0x01}), true, false).is_discarded()); } + SECTION("invalid UTF-8 in string (see #5529)") + { + // a fixstr of length 2 (0xA0 | 2) whose bytes are not valid UTF-8 + // (0xC0 0xAE is an overlong encoding of '.') must be rejected at + // decode time, matching every other kind of malformed binary + // input, rather than only failing later when the resulting + // value is dumped + json _; + CHECK_THROWS_WITH_AS(_ = json::from_msgpack(std::vector({0xa2, 0xc0, 0xae})), "[json.exception.parse_error.113] parse error at byte 3: syntax error while parsing MessagePack string: invalid string: ill-formed UTF-8 byte", json::parse_error&); + CHECK(json::from_msgpack(std::vector({0xa2, 0xc0, 0xae}), true, false).is_discarded()); + + // a MessagePack bin8 blob with the very same bytes is NOT text + // and must still be accepted as-is + CHECK_NOTHROW(_ = json::from_msgpack(std::vector({0xc4, 0x02, 0xc0, 0xae}))); + CHECK(_ == json::binary(std::vector({0xc0, 0xae}))); + + // valid UTF-8 must still round-trip + const json j = "h\xc3\xa9llo, w\xc3\xb6rld! \xe6\x97\xa5\xe6\x9c\xac\xe8\xaa\x9e"; // héllo, wörld! 日本語 + CHECK(json::from_msgpack(json::to_msgpack(j)) == j); + } + SECTION("strict mode") { std::vector const vec = {0xc0, 0xc0}; diff --git a/tests/src/unit-serialization.cpp b/tests/src/unit-serialization.cpp index 00b305a75..45617c3b3 100644 --- a/tests/src/unit-serialization.cpp +++ b/tests/src/unit-serialization.cpp @@ -94,6 +94,16 @@ TEST_CASE("serialization") CHECK(j.dump(-1, ' ', true, json::error_handler_t::replace) == "\"\\u00e4\\ufffd\\u00fc\""); } + SECTION("invalid character (regression guard for shared UTF-8 decoder, see #5529)") + { + // dump_escaped_impl() now calls the UTF-8 decoder shared with the + // binary readers (detail::decode() in string_utils.hpp) instead + // of a private copy; the exact type_error.316 message/behavior + // must stay byte-for-byte the same as before that extraction + const json j = "ä\xA9ü"; + CHECK_THROWS_WITH_AS(utils::ignore_return_value(j.dump()), "[json.exception.type_error.316] invalid UTF-8 byte at index 2: 0xA9", json::type_error&); + } + SECTION("ending with incomplete character") { const json j = "123\xC2"; From c60a0bc336cf546606a232e0af0b24720e0e893a Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 25 Sep 2026 21:53:49 +0200 Subject: [PATCH 51/64] Allocate the deep copy's key scratch space with the provided allocator (#5573) * Allocate the deep copy's key scratch space with the provided allocator The iterative deep copy builds each object's keys in a temporary vector of key/value pairs before handing them to the object's range constructor. That vector holds basic_json values, so like the values themselves it now uses AllocatorType instead of std::allocator. Also document that AllocatorType covers the JSON values, while most temporary storage still uses std::allocator. Signed-off-by: Niels Lohmann * Count allocate_at_least in the scratch-counting test allocator From C++23 on, libc++'s containers allocate through allocate_at_least when the allocator has one. The test allocator inherited it from std::allocator, so the scratch allocations were not counted and the test failed on Xcode. Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- .../features/types/template_parameters.md | 6 +- include/nlohmann/json.hpp | 3 +- single_include/nlohmann/json.hpp | 3 +- tests/src/unit-allocator.cpp | 76 +++++++++++++++++++ 4 files changed, 85 insertions(+), 3 deletions(-) diff --git a/docs/mkdocs/docs/features/types/template_parameters.md b/docs/mkdocs/docs/features/types/template_parameters.md index 760972dff..2a670f81c 100644 --- a/docs/mkdocs/docs/features/types/template_parameters.md +++ b/docs/mkdocs/docs/features/types/template_parameters.md @@ -562,7 +562,11 @@ binary32 or binary64 field and have no encoding for `#!cpp long double`. ## `AllocatorType` `AllocatorType` is instantiated with **one** argument, for each of `object_t`, `array_t`, `string_t`, `binary_t`, -`basic_json`, and `#!cpp std::pair`. +`basic_json`, `#!cpp std::pair`, and `#!cpp std::pair`. + +`AllocatorType` is not the only allocator a `basic_json` uses. It allocates the JSON values themselves, but most +temporary storage is allocated with `#!cpp std::allocator`. This includes the parser's stacks and the stacks that +process deeply nested values without recursion. ### Always required diff --git a/include/nlohmann/json.hpp b/include/nlohmann/json.hpp index 1090ed105..c578fd11a 100644 --- a/include/nlohmann/json.hpp +++ b/include/nlohmann/json.hpp @@ -1003,7 +1003,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec using copy_worklist_t = std::vector>; /// scratch space to build the key skeleton of an object copy in one go - using copy_scratch_t = std::vector>; + using copy_scratch_value_t = std::pair; + using copy_scratch_t = std::vector>; /// @brief copy everything of @a src into @a dst but its type and value static void copy_metadata(const basic_json& src, basic_json& dst) diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 647312562..b92644c40 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -25835,7 +25835,8 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec using copy_worklist_t = std::vector>; /// scratch space to build the key skeleton of an object copy in one go - using copy_scratch_t = std::vector>; + using copy_scratch_value_t = std::pair; + using copy_scratch_t = std::vector>; /// @brief copy everything of @a src into @a dst but its type and value static void copy_metadata(const basic_json& src, basic_json& dst) diff --git a/tests/src/unit-allocator.cpp b/tests/src/unit-allocator.cpp index 5c7b4230f..cdc532e3a 100644 --- a/tests/src/unit-allocator.cpp +++ b/tests/src/unit-allocator.cpp @@ -270,6 +270,82 @@ TEST_CASE("controlled bad_alloc") } } +namespace +{ +// counts the allocations of pairs with a non-const first member: the object +// types store std::pair, so only the scratch space of the +// iterative deep copy allocates std::pair +std::size_t scratch_pair_allocations = 0; + +template +struct is_scratch_pair : std::false_type {}; + +template +struct is_scratch_pair> : std::integral_constant < bool, !std::is_const::value > {}; + +template +struct scratch_counting_allocator : std::allocator +{ + using std::allocator::allocator; + + T* allocate(std::size_t n) + { + if (is_scratch_pair::value) + { + ++scratch_pair_allocations; + } + return std::allocator::allocate(n); + } + +#ifdef __cpp_lib_allocate_at_least + // std::allocator::allocate_at_least would bypass the counting, and + // libc++'s containers prefer it over allocate from C++23 on + auto allocate_at_least(std::size_t n) + { + if (is_scratch_pair::value) + { + ++scratch_pair_allocations; + } + return std::allocator::allocate_at_least(n); + } +#endif + + template + struct rebind + { + using other = scratch_counting_allocator; + }; +}; +} // namespace + +TEST_CASE("deep copy uses the provided allocator") +{ + using counting_json = nlohmann::basic_json; + + // deeper than the 128 levels the copy constructor descends into, so the + // innermost objects are copied by the iterative deep copy + counting_json j = 1; + for (std::size_t i = 0; i < 300; ++i) + { + counting_json wrapper = counting_json::object(); + wrapper["a"] = std::move(j); + j = std::move(wrapper); + } + + scratch_pair_allocations = 0; + // NOLINTNEXTLINE(performance-unnecessary-copy-initialization): the copy is what is tested + const counting_json copy(j); + CHECK(scratch_pair_allocations > 0); + CHECK(copy == j); +} + namespace { template From 1e442620917d70322f74896cf16cf8a8be2e66c8 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 25 Sep 2026 21:54:11 +0200 Subject: [PATCH 52/64] Make JSON_STRICT_NUL_HANDLING part of the ABI tag (#5560) * Make JSON_STRICT_NUL_HANDLING part of the ABI tag JSON_STRICT_NUL_HANDLING (#5534) changes the bodies of inline functions: the lexer's handling of '\0' and input_adapter() for char arrays. So translation units compiled with and without it define the same functions differently, an ODR violation - the case the ABI tag exists for, as with JSON_BRACE_INIT_COPY_SEMANTICS (_bics). It now appends _snul to the inline namespace. The macro is new in 3.13.0, so no existing namespace changes. Its default moves to abi_macros.hpp, and it is only #undef'd without JSON_TEST_KEEP_MACROS, as for the other ABI macros. The ABI config tests, the namespace docs, the macro's docs and the Natvis file cover the new tag. Signed-off-by: Niels Lohmann * Amalgamate Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- .../api/macros/json_strict_nul_handling.md | 6 + docs/mkdocs/docs/features/namespace.md | 1 + include/nlohmann/detail/abi_macros.hpp | 19 +- include/nlohmann/detail/macro_scope.hpp | 4 - include/nlohmann/detail/macro_unscope.hpp | 2 +- nlohmann_json.natvis | 1920 +++++++++++++++++ single_include/nlohmann/json.hpp | 25 +- single_include/nlohmann/json_fwd.hpp | 19 +- tests/abi/config/default.cpp | 4 + tests/abi/config/noversion.cpp | 4 + tests/src/unit-class_parser.cpp | 16 +- tools/generate_natvis/generate_natvis.py | 2 +- 12 files changed, 1998 insertions(+), 24 deletions(-) diff --git a/docs/mkdocs/docs/api/macros/json_strict_nul_handling.md b/docs/mkdocs/docs/api/macros/json_strict_nul_handling.md index 1832f2353..a50ce2645 100644 --- a/docs/mkdocs/docs/api/macros/json_strict_nul_handling.md +++ b/docs/mkdocs/docs/api/macros/json_strict_nul_handling.md @@ -65,6 +65,12 @@ The default value is `0` (disabled — existing behavior is preserved). for CBOR or MessagePack, are never affected by this trimming; their full extent - including a genuine trailing `0x00` - is always preserved, in both states of this macro. +!!! note "ABI compatibility" + + The value of this macro is encoded in the [namespace](../../features/namespace.md) (tag `_snul`), resulting in + distinct symbol names. Translation units compiled with and without it can therefore be linked into the same program + without One Definition Rule (ODR) violations, but they cannot exchange instances of library types. + !!! tip "Workaround without the macro" To reject a NUL byte without enabling this macro, trim your input yourself before calling `parse()`: diff --git a/docs/mkdocs/docs/features/namespace.md b/docs/mkdocs/docs/features/namespace.md index 5eb4a76a9..577f5e221 100644 --- a/docs/mkdocs/docs/features/namespace.md +++ b/docs/mkdocs/docs/features/namespace.md @@ -19,6 +19,7 @@ The complete default namespace name is derived as follows: - [`JSON_BRACE_INIT_COPY_SEMANTICS`](../api/macros/json_brace_init_copy_semantics.md) defined non-zero appends `_bics`. - [`JSON_PRECISE_STREAM_POSITION`](../api/macros/json_precise_stream_position.md) defined non-zero appends `_psp`. + - [`JSON_STRICT_NUL_HANDLING`](../api/macros/json_strict_nul_handling.md) defined non-zero appends `_snul`. - The inline namespace ends with the suffix `_v` followed by the 3 components of the version number separated by underscores. To omit the version component, see [Disabling the version component](#disabling-the-version-component) below. diff --git a/include/nlohmann/detail/abi_macros.hpp b/include/nlohmann/detail/abi_macros.hpp index a6666c66e..0bace616a 100644 --- a/include/nlohmann/detail/abi_macros.hpp +++ b/include/nlohmann/detail/abi_macros.hpp @@ -42,6 +42,10 @@ #define JSON_PRECISE_STREAM_POSITION 0 #endif +#ifndef JSON_STRICT_NUL_HANDLING + #define JSON_STRICT_NUL_HANDLING 0 +#endif + #if JSON_DIAGNOSTICS #define NLOHMANN_JSON_ABI_TAG_DIAGNOSTICS _diag #else @@ -72,14 +76,20 @@ #define NLOHMANN_JSON_ABI_TAG_PRECISE_STREAM_POSITION #endif +#if JSON_STRICT_NUL_HANDLING + #define NLOHMANN_JSON_ABI_TAG_STRICT_NUL_HANDLING _snul +#else + #define NLOHMANN_JSON_ABI_TAG_STRICT_NUL_HANDLING +#endif + #ifndef NLOHMANN_JSON_NAMESPACE_NO_VERSION #define NLOHMANN_JSON_NAMESPACE_NO_VERSION 0 #endif // Construct the namespace ABI tags component -#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e) json_abi ## a ## b ## c ## d ## e -#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d, e) \ - NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e) +#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f) json_abi ## a ## b ## c ## d ## e ## f +#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d, e, f) \ + NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f) #define NLOHMANN_JSON_ABI_TAGS \ NLOHMANN_JSON_ABI_TAGS_CONCAT( \ @@ -87,7 +97,8 @@ NLOHMANN_JSON_ABI_TAG_LEGACY_DISCARDED_VALUE_COMPARISON, \ NLOHMANN_JSON_ABI_TAG_DIAGNOSTIC_POSITIONS, \ NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS, \ - NLOHMANN_JSON_ABI_TAG_PRECISE_STREAM_POSITION) + NLOHMANN_JSON_ABI_TAG_PRECISE_STREAM_POSITION, \ + NLOHMANN_JSON_ABI_TAG_STRICT_NUL_HANDLING) // Construct the namespace version component #define NLOHMANN_JSON_NAMESPACE_VERSION_CONCAT_EX(major, minor, patch) \ diff --git a/include/nlohmann/detail/macro_scope.hpp b/include/nlohmann/detail/macro_scope.hpp index 9094aa6e4..3a4eb79f4 100644 --- a/include/nlohmann/detail/macro_scope.hpp +++ b/include/nlohmann/detail/macro_scope.hpp @@ -919,7 +919,3 @@ void templated_json_throw(ExceptionType exception) #ifndef JSON_USE_GLOBAL_UDLS #define JSON_USE_GLOBAL_UDLS 1 #endif - -#ifndef JSON_STRICT_NUL_HANDLING - #define JSON_STRICT_NUL_HANDLING 0 -#endif diff --git a/include/nlohmann/detail/macro_unscope.hpp b/include/nlohmann/detail/macro_unscope.hpp index 6ace4cf4a..a73951a22 100644 --- a/include/nlohmann/detail/macro_unscope.hpp +++ b/include/nlohmann/detail/macro_unscope.hpp @@ -26,7 +26,6 @@ #undef JSON_NO_UNIQUE_ADDRESS #undef JSON_DISABLE_ENUM_SERIALIZATION #undef JSON_USE_GLOBAL_UDLS -#undef JSON_STRICT_NUL_HANDLING #ifndef JSON_TEST_KEEP_MACROS #undef JSON_CATCH @@ -46,6 +45,7 @@ #undef JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON #undef JSON_BRACE_INIT_COPY_SEMANTICS #undef JSON_PRECISE_STREAM_POSITION + #undef JSON_STRICT_NUL_HANDLING #endif #include diff --git a/nlohmann_json.natvis b/nlohmann_json.natvis index 8f4eec31a..ed443145e 100644 --- a/nlohmann_json.natvis +++ b/nlohmann_json.natvis @@ -395,6 +395,66 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -635,6 +695,66 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -815,6 +935,66 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -935,6 +1115,66 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -995,6 +1235,126 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -1175,6 +1535,66 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -1295,6 +1715,66 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -1355,6 +1835,126 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -1475,6 +2075,66 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -1535,6 +2195,126 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -1595,6 +2375,186 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -1715,6 +2675,66 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -1775,6 +2795,126 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -1835,6 +2975,186 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -1895,6 +3215,246 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + null @@ -1955,4 +3515,364 @@ + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + + + + null + {*(m_data.m_value.object)} + {*(m_data.m_value.array)} + {*(m_data.m_value.string)} + {m_data.m_value.boolean} + {m_data.m_value.number_integer} + {m_data.m_value.number_unsigned} + {m_data.m_value.number_float} + discarded + + + *(m_data.m_value.object),view(simple) + + + *(m_data.m_value.array),view(simple) + + + + + + + {second} + + second + + + diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index b92644c40..ac80ddc5a 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -99,6 +99,10 @@ #define JSON_PRECISE_STREAM_POSITION 0 #endif +#ifndef JSON_STRICT_NUL_HANDLING + #define JSON_STRICT_NUL_HANDLING 0 +#endif + #if JSON_DIAGNOSTICS #define NLOHMANN_JSON_ABI_TAG_DIAGNOSTICS _diag #else @@ -129,14 +133,20 @@ #define NLOHMANN_JSON_ABI_TAG_PRECISE_STREAM_POSITION #endif +#if JSON_STRICT_NUL_HANDLING + #define NLOHMANN_JSON_ABI_TAG_STRICT_NUL_HANDLING _snul +#else + #define NLOHMANN_JSON_ABI_TAG_STRICT_NUL_HANDLING +#endif + #ifndef NLOHMANN_JSON_NAMESPACE_NO_VERSION #define NLOHMANN_JSON_NAMESPACE_NO_VERSION 0 #endif // Construct the namespace ABI tags component -#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e) json_abi ## a ## b ## c ## d ## e -#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d, e) \ - NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e) +#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f) json_abi ## a ## b ## c ## d ## e ## f +#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d, e, f) \ + NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f) #define NLOHMANN_JSON_ABI_TAGS \ NLOHMANN_JSON_ABI_TAGS_CONCAT( \ @@ -144,7 +154,8 @@ NLOHMANN_JSON_ABI_TAG_LEGACY_DISCARDED_VALUE_COMPARISON, \ NLOHMANN_JSON_ABI_TAG_DIAGNOSTIC_POSITIONS, \ NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS, \ - NLOHMANN_JSON_ABI_TAG_PRECISE_STREAM_POSITION) + NLOHMANN_JSON_ABI_TAG_PRECISE_STREAM_POSITION, \ + NLOHMANN_JSON_ABI_TAG_STRICT_NUL_HANDLING) // Construct the namespace version component #define NLOHMANN_JSON_NAMESPACE_VERSION_CONCAT_EX(major, minor, patch) \ @@ -3320,10 +3331,6 @@ void templated_json_throw(ExceptionType exception) #define JSON_USE_GLOBAL_UDLS 1 #endif -#ifndef JSON_STRICT_NUL_HANDLING - #define JSON_STRICT_NUL_HANDLING 0 -#endif - #if JSON_HAS_THREE_WAY_COMPARISON #include // partial_ordering #endif @@ -31413,7 +31420,6 @@ struct formatter // NOLINT(cert-dcl58-c #undef JSON_NO_UNIQUE_ADDRESS #undef JSON_DISABLE_ENUM_SERIALIZATION #undef JSON_USE_GLOBAL_UDLS -#undef JSON_STRICT_NUL_HANDLING #ifndef JSON_TEST_KEEP_MACROS #undef JSON_CATCH @@ -31433,6 +31439,7 @@ struct formatter // NOLINT(cert-dcl58-c #undef JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON #undef JSON_BRACE_INIT_COPY_SEMANTICS #undef JSON_PRECISE_STREAM_POSITION + #undef JSON_STRICT_NUL_HANDLING #endif // #include diff --git a/single_include/nlohmann/json_fwd.hpp b/single_include/nlohmann/json_fwd.hpp index af776d652..f23ea4820 100644 --- a/single_include/nlohmann/json_fwd.hpp +++ b/single_include/nlohmann/json_fwd.hpp @@ -60,6 +60,10 @@ #define JSON_PRECISE_STREAM_POSITION 0 #endif +#ifndef JSON_STRICT_NUL_HANDLING + #define JSON_STRICT_NUL_HANDLING 0 +#endif + #if JSON_DIAGNOSTICS #define NLOHMANN_JSON_ABI_TAG_DIAGNOSTICS _diag #else @@ -90,14 +94,20 @@ #define NLOHMANN_JSON_ABI_TAG_PRECISE_STREAM_POSITION #endif +#if JSON_STRICT_NUL_HANDLING + #define NLOHMANN_JSON_ABI_TAG_STRICT_NUL_HANDLING _snul +#else + #define NLOHMANN_JSON_ABI_TAG_STRICT_NUL_HANDLING +#endif + #ifndef NLOHMANN_JSON_NAMESPACE_NO_VERSION #define NLOHMANN_JSON_NAMESPACE_NO_VERSION 0 #endif // Construct the namespace ABI tags component -#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e) json_abi ## a ## b ## c ## d ## e -#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d, e) \ - NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e) +#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f) json_abi ## a ## b ## c ## d ## e ## f +#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c, d, e, f) \ + NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c, d, e, f) #define NLOHMANN_JSON_ABI_TAGS \ NLOHMANN_JSON_ABI_TAGS_CONCAT( \ @@ -105,7 +115,8 @@ NLOHMANN_JSON_ABI_TAG_LEGACY_DISCARDED_VALUE_COMPARISON, \ NLOHMANN_JSON_ABI_TAG_DIAGNOSTIC_POSITIONS, \ NLOHMANN_JSON_ABI_TAG_BRACE_INIT_COPY_SEMANTICS, \ - NLOHMANN_JSON_ABI_TAG_PRECISE_STREAM_POSITION) + NLOHMANN_JSON_ABI_TAG_PRECISE_STREAM_POSITION, \ + NLOHMANN_JSON_ABI_TAG_STRICT_NUL_HANDLING) // Construct the namespace version component #define NLOHMANN_JSON_NAMESPACE_VERSION_CONCAT_EX(major, minor, patch) \ diff --git a/tests/abi/config/default.cpp b/tests/abi/config/default.cpp index 8f66dfbf1..879322dd0 100644 --- a/tests/abi/config/default.cpp +++ b/tests/abi/config/default.cpp @@ -40,6 +40,10 @@ TEST_CASE("default namespace") expected += "_psp"; #endif +#if JSON_STRICT_NUL_HANDLING + expected += "_snul"; +#endif + expected += "_v" STRINGIZE(NLOHMANN_JSON_VERSION_MAJOR); expected += "_" STRINGIZE(NLOHMANN_JSON_VERSION_MINOR); expected += "_" STRINGIZE(NLOHMANN_JSON_VERSION_PATCH) "::basic_json"; diff --git a/tests/abi/config/noversion.cpp b/tests/abi/config/noversion.cpp index 12b7603b9..4b1eb6ee4 100644 --- a/tests/abi/config/noversion.cpp +++ b/tests/abi/config/noversion.cpp @@ -41,6 +41,10 @@ TEST_CASE("default namespace without version component") expected += "_psp"; #endif +#if JSON_STRICT_NUL_HANDLING + expected += "_snul"; +#endif + expected += "::basic_json"; // fallback for Clang diff --git a/tests/src/unit-class_parser.cpp b/tests/src/unit-class_parser.cpp index e532992f4..707f23c18 100644 --- a/tests/src/unit-class_parser.cpp +++ b/tests/src/unit-class_parser.cpp @@ -11,11 +11,15 @@ // capture whether JSON_STRICT_NUL_HANDLING was enabled on the command line // (e.g. -DJSON_STRICT_NUL_HANDLING=1) *before* including json.hpp, since the // library #undefs JSON_STRICT_NUL_HANDLING itself once the header has been -// fully processed (see include/nlohmann/detail/macro_unscope.hpp) +// fully processed unless JSON_TEST_KEEP_MACROS is defined (see +// include/nlohmann/detail/macro_unscope.hpp) #if defined(JSON_STRICT_NUL_HANDLING) && (JSON_STRICT_NUL_HANDLING == 1) #define JSON_TEST_STRICT_NUL_HANDLING_ENABLED 1 #endif +#define JSON_TEST_STRINGIZE_EX(x) #x +#define JSON_TEST_STRINGIZE(x) JSON_TEST_STRINGIZE_EX(x) + #define JSON_TESTS_PRIVATE #include using nlohmann::json; @@ -566,6 +570,16 @@ TEST_CASE("parser class") // left at its default or forced to 1 (e.g. by the dedicated // ci_test_strict_nul_handling CI target), so only the section // matching the actual, compiled-in behavior can pass. + SECTION("the macro is part of the ABI tag") + { + const std::string ns = JSON_TEST_STRINGIZE(NLOHMANN_JSON_NAMESPACE); +#if defined(JSON_TEST_STRICT_NUL_HANDLING_ENABLED) + CHECK(ns.find("_snul") != std::string::npos); +#else + CHECK(ns.find("_snul") == std::string::npos); +#endif + } + #if !defined(JSON_TEST_STRICT_NUL_HANDLING_ENABLED) SECTION("default behavior (macro not enabled)") { diff --git a/tools/generate_natvis/generate_natvis.py b/tools/generate_natvis/generate_natvis.py index fb1210db1..8562e8f1a 100755 --- a/tools/generate_natvis/generate_natvis.py +++ b/tools/generate_natvis/generate_natvis.py @@ -20,7 +20,7 @@ if __name__ == '__main__': namespaces = ['nlohmann'] abi_prefix = 'json_abi' - abi_tags = ['_diag', '_ldvcmp', '_dp', '_bics', '_psp'] + abi_tags = ['_diag', '_ldvcmp', '_dp', '_bics', '_psp', '_snul'] version = '_v' + args.version.replace('.', '_') inline_namespaces = [] From 95e9a5931c828cd7306eff36090a5b38e0e3f598 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Fri, 25 Sep 2026 21:56:39 +0200 Subject: [PATCH 53/64] Write BSON in linear time, without recursing per nesting level (#5553) * Write BSON in linear time, without recursing per nesting level to_bson() had two problems with nested values: - It recursed once per nesting level, so a value nested deeply enough - 100,000 levels on an 8 MiB stack - exhausted the call stack and terminated the process, although parse() accepts such values without complaint. - BSON prefixes every document and array with its length. The writer computed that length by walking the entire value below it, again for every nested document it wrote, which made serializing O(size x depth). A 200-level document took 30 ms instead of 1. Both passes are now iterative, and each length is computed exactly once: - calc_bson_sizes() computes the length of every document and array in one pass, each from the lengths of its entries, into a table ordered the way they are written. - write_bson_document() then writes the document, taking each length from the table. Everything observable is unchanged, as a differential test against develop confirms byte for byte: - The same bytes are written. - A key containing U+0000 still throws out_of_range.409 for the same first key, with the same diagnostics path, before anything is written. - A document too large for BSON still throws out_of_range.412 before anything is written. - A binary subtype above 255 still throws out_of_range.415 after the same partial output. Only the enclosing objects and arrays are kept on a stack, so a flat document allocates nothing for it. Measured against develop (clang -O3, median of 201 runs): flat objects unchanged, flat arrays 37% faster (the array length was computed twice), a nested 3,000-object document 2x faster, a 200-level document 33x faster. to_bson.md documented the quadratic complexity since #5334; it is linear again. Fixes #5392 for BSON, and #5308. Signed-off-by: Niels Lohmann * Do not require a default-constructible string_t in the BSON writer GCC 4.9 and MSVC rejected the test's huge_string_t, which has no default constructor; develop never default-constructed string_t here either. Signed-off-by: Niels Lohmann * Let the BSON index-name helper only fill its output parameter It returned a reference to the string it filled, so callers held a second name for index_name. Addresses review feedback. Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- docs/mkdocs/docs/api/basic_json/to_bson.md | 6 +- .../nlohmann/detail/output/binary_writer.hpp | 325 ++++++++++++------ single_include/nlohmann/json.hpp | 325 ++++++++++++------ tests/src/unit-bson.cpp | 66 +++- 4 files changed, 512 insertions(+), 210 deletions(-) diff --git a/docs/mkdocs/docs/api/basic_json/to_bson.md b/docs/mkdocs/docs/api/basic_json/to_bson.md index 786fbc9e0..4cd45a57d 100644 --- a/docs/mkdocs/docs/api/basic_json/to_bson.md +++ b/docs/mkdocs/docs/api/basic_json/to_bson.md @@ -46,9 +46,8 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va ## Complexity -Proportional to the size of the JSON value `j` multiplied by its maximum nesting -depth, `O(n × d)`. BSON length prefixes are computed recursively before nested -values are written. +Linear in the size of the JSON value `j`. The length prefixes of all nested documents and arrays are computed in one +pass before anything is written. ## Examples @@ -77,3 +76,4 @@ values are written. ## Version history - Added in version 3.4.0. +- Linear in the size of `j`, and no longer limited by the call stack for deeply nested values, since version 3.13.0. diff --git a/include/nlohmann/detail/output/binary_writer.hpp b/include/nlohmann/detail/output/binary_writer.hpp index eaa8d63b6..5a0722a18 100644 --- a/include/nlohmann/detail/output/binary_writer.hpp +++ b/include/nlohmann/detail/output/binary_writer.hpp @@ -122,7 +122,7 @@ class binary_writer { case value_t::object: { - write_bson_object(*j.m_data.m_value.object); + write_bson_document(j); break; } @@ -1197,35 +1197,6 @@ class binary_writer } } - /*! - @brief Writes a BSON element with key @a name and object @a value - */ - void write_bson_object_entry(const string_t& name, - const typename BasicJsonType::object_t& value) - { - write_bson_entry_header(name, 0x03); // object - write_bson_object(value); - } - - /*! - @return The size of the BSON-encoded array @a value - */ - static std::size_t calc_bson_array_size(const typename BasicJsonType::array_t& value) - { - std::size_t array_index = 0ul; - - const std::size_t embedded_document_size = std::accumulate(std::begin(value), std::end(value), static_cast(0), [&array_index](std::size_t result, const typename BasicJsonType::array_t::value_type & el) - { - // the index is built as a std::string, while calc_bson_element_size - // takes a string_t; convert explicitly, as the two are only - // implicitly convertible for some string types - const auto key = std::to_string(array_index++); - return result + calc_bson_element_size(string_t(key.data(), key.size()), el); - }); - - return sizeof(std::int32_t) + embedded_document_size + 1ul; - } - /*! @return The size of the BSON-encoded binary array @a value */ @@ -1234,29 +1205,6 @@ class binary_writer return sizeof(std::int32_t) + value.size() + 1ul; } - /*! - @brief Writes a BSON element with key @a name and array @a value - */ - void write_bson_array(const string_t& name, - const typename BasicJsonType::array_t& value) - { - write_bson_entry_header(name, 0x04); // array - write_number(to_bson_length(calc_bson_array_size(value)), true); - - std::size_t array_index = 0ul; - - for (const auto& el : value) - { - // the index is built as a std::string, while write_bson_element takes - // a string_t; convert explicitly, as the two are only implicitly - // convertible for some string types - const auto key = std::to_string(array_index++); - write_bson_element(string_t(key.data(), key.size()), el); - } - - oa.write_character(to_char_type(0x00)); - } - /*! @brief Writes a BSON element with key @a name and binary value @a value */ @@ -1278,43 +1226,37 @@ class binary_writer } /*! - @brief Calculates the size necessary to serialize the JSON value @a j with its @a name - @return The calculated size for the BSON document entry for @a j with the given @a name. + @return The size of the value of the BSON document entry for @a j, which + is neither an object nor an array */ - static std::size_t calc_bson_element_size(const string_t& name, - const BasicJsonType& j) + static std::size_t calc_bson_value_size(const BasicJsonType& j) { - const auto header_size = calc_bson_entry_header_size(name, j); switch (j.type()) { - case value_t::object: - return header_size + calc_bson_object_size(*j.m_data.m_value.object); - - case value_t::array: - return header_size + calc_bson_array_size(*j.m_data.m_value.array); - case value_t::binary: - return header_size + calc_bson_binary_size(*j.m_data.m_value.binary); + return calc_bson_binary_size(*j.m_data.m_value.binary); case value_t::boolean: - return header_size + 1ul; + return 1ul; case value_t::number_float: - return header_size + 8ul; + return 8ul; case value_t::number_integer: - return header_size + calc_bson_integer_size(j.m_data.m_value.number_integer); + return calc_bson_integer_size(j.m_data.m_value.number_integer); case value_t::number_unsigned: - return header_size + calc_bson_unsigned_size(j.m_data.m_value.number_unsigned); + return calc_bson_unsigned_size(j.m_data.m_value.number_unsigned); case value_t::string: - return header_size + calc_bson_string_size(*j.m_data.m_value.string); + return calc_bson_string_size(*j.m_data.m_value.string); case value_t::null: - return header_size + 0ul; + return 0ul; // LCOV_EXCL_START + case value_t::object: + case value_t::array: case value_t::discarded: default: JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) @@ -1324,22 +1266,13 @@ class binary_writer } /*! - @brief Serializes the JSON value @a j to BSON and associates it with the - key @a name. - @param name The name to associate with the JSON entity @a j within the - current BSON document + @brief Writes the BSON document entry with key @a name for @a j, which is + neither an object nor an array */ - void write_bson_element(const string_t& name, - const BasicJsonType& j) + void write_bson_value(const string_t& name, const BasicJsonType& j) { switch (j.type()) { - case value_t::object: - return write_bson_object_entry(name, *j.m_data.m_value.object); - - case value_t::array: - return write_bson_array(name, *j.m_data.m_value.array); - case value_t::binary: return write_bson_binary(name, *j.m_data.m_value.binary); @@ -1362,6 +1295,8 @@ class binary_writer return write_bson_null(name); // LCOV_EXCL_START + case value_t::object: + case value_t::array: case value_t::discarded: default: JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) @@ -1370,37 +1305,221 @@ class binary_writer } } - /*! - @brief Calculates the size of the BSON serialization of the given - JSON-object @a j. - @param[in] value JSON value to serialize - @pre value.type() == value_t::object - */ - static std::size_t calc_bson_object_size(const typename BasicJsonType::object_t& value) + /// @brief an object or array of the BSON document being sized or written + struct bson_frame { - const std::size_t document_size = std::accumulate(value.begin(), value.end(), static_cast(0), - [](size_t result, const typename BasicJsonType::object_t::value_type & el) + explicit bson_frame(const BasicJsonType* value_, const std::size_t size_slot_ = 0) + : value(value_) + , size_slot(size_slot_) { - return result += calc_bson_element_size(el.first, el.second); - }); + if (value->is_object()) + { + member = value->m_data.m_value.object->cbegin(); + } + } - return sizeof(std::int32_t) + document_size + 1ul; + /// the object or array + const BasicJsonType* value; + /// objects: the next member + typename BasicJsonType::object_t::const_iterator member{}; + /// arrays: the index of the next element + std::size_t index = 0; + /// @ref calc_bson_sizes only: where its size goes in the table + std::size_t size_slot; + /// @ref calc_bson_sizes only: the size of its entries seen so far + std::size_t entries_size = 0; + }; + + /*! + @brief creates the name BSON gives the array element with index @a index + @param[out] name receives the decimal index + */ + static void create_bson_index_name(const std::size_t index, string_t& name) + { + // the index is built as a std::string; convert explicitly, as the + // two are only implicitly convertible for some string types + const auto key = std::to_string(index); + name = string_t(key.data(), key.size()); } /*! - @param[in] value JSON value to serialize - @pre value.type() == value_t::object + @brief Calculates the size of every object and array in the BSON document + @a document, including the document itself. + + BSON prefixes every document and array with its size, so all of them have + to be known before the first byte is written. They are computed in a + single pass, each one from the sizes of its entries, which keeps + serializing linear in the size of the document; computing each size by + walking the entire value below it made it quadratic in the nesting depth. + The pass keeps the objects and arrays it has entered on an explicit stack, + so a deeply nested value cannot exhaust the call stack. + + @param[in] document the JSON object to serialize + @param[out] nested_sizes the sizes of the objects and arrays in + @a document, in the order they are written + @return the size of @a document + @throw out_of_range.409 if a key contains U+0000, before anything is + written */ - void write_bson_object(const typename BasicJsonType::object_t& value) + static std::size_t calc_bson_sizes(const BasicJsonType& document, std::vector& nested_sizes) { - write_number(to_bson_length(calc_bson_object_size(value)), true); + // the object or array whose entries are being sized, and the ones it + // is in; nothing is allocated unless the document nests + bson_frame current(&document); + std::vector parents; + // string_t need not be default constructible + string_t index_name("", 0); - for (const auto& el : value) + while (true) { - write_bson_element(el.first, el.second); - } + // size entries until the current object or array is done, or an + // entry is an object or array itself + const BasicJsonType* nested = nullptr; + if (current.value->is_object()) + { + const auto& object = *current.value->m_data.m_value.object; + while (nested == nullptr && current.member != object.cend()) + { + const auto& el = *current.member; + ++current.member; + current.entries_size += calc_bson_entry_header_size(el.first, el.second); + if (el.second.is_structured()) + { + nested = &el.second; + } + else + { + current.entries_size += calc_bson_value_size(el.second); + } + } + } + else + { + const auto& array = *current.value->m_data.m_value.array; + while (nested == nullptr && current.index < array.size()) + { + const BasicJsonType& el = array[current.index]; + create_bson_index_name(current.index, index_name); + current.entries_size += calc_bson_entry_header_size(index_name, el); + ++current.index; + if (el.is_structured()) + { + nested = ⪙ + } + else + { + current.entries_size += calc_bson_value_size(el); + } + } + } - oa.write_character(to_char_type(0x00)); + if (nested != nullptr) + { + // its size is added to the current one's once it is done + nested_sizes.push_back(0); + parents.push_back(std::move(current)); + current = bson_frame(nested, nested_sizes.size() - 1); + continue; + } + + // the int32 size, the entries, and the terminating null byte + const std::size_t size = sizeof(std::int32_t) + current.entries_size + 1ul; + if (parents.empty()) + { + return size; + } + nested_sizes[current.size_slot] = size; + current = std::move(parents.back()); + parents.pop_back(); + current.entries_size += size; + } + } + + /*! + @brief Serializes the JSON object @a document as a BSON document + + Writes the objects and arrays in it without the call stack, keeping the + ones it has entered on an explicit stack, so a deeply nested value + cannot exhaust the call stack. + + @param[in] document the JSON object to serialize + @pre document.type() == value_t::object + */ + void write_bson_document(const BasicJsonType& document) + { + std::vector nested_sizes; + const std::size_t document_size = calc_bson_sizes(document, nested_sizes); + write_number(to_bson_length(document_size), true); + + // the object or array whose entries are being written, and the ones + // it is in + bson_frame current(&document); + std::vector parents; + std::size_t next_size = 0; + // string_t need not be default constructible + string_t index_name("", 0); + + while (true) + { + // write entries until the current object or array is done, or an + // entry is an object or array itself + const string_t* nested_name = nullptr; + const BasicJsonType* nested = nullptr; + if (current.value->is_object()) + { + const auto& object = *current.value->m_data.m_value.object; + while (nested == nullptr && current.member != object.cend()) + { + const auto& el = *current.member; + ++current.member; + if (el.second.is_structured()) + { + nested_name = &el.first; + nested = &el.second; + } + else + { + write_bson_value(el.first, el.second); + } + } + } + else + { + const auto& array = *current.value->m_data.m_value.array; + while (nested == nullptr && current.index < array.size()) + { + const BasicJsonType& el = array[current.index]; + create_bson_index_name(current.index, index_name); + ++current.index; + if (el.is_structured()) + { + nested_name = &index_name; + nested = ⪙ + } + else + { + write_bson_value(index_name, el); + } + } + } + + if (nested != nullptr) + { + write_bson_entry_header(*nested_name, nested->is_object() ? 0x03 : 0x04); + write_number(to_bson_length(nested_sizes[next_size++]), true); + parents.push_back(std::move(current)); + current = bson_frame(nested); + continue; + } + + oa.write_character(to_char_type(0x00)); + if (parents.empty()) + { + return; + } + current = std::move(parents.back()); + parents.pop_back(); + } } ////////// diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index ac80ddc5a..4981c3b02 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -19587,7 +19587,7 @@ class binary_writer { case value_t::object: { - write_bson_object(*j.m_data.m_value.object); + write_bson_document(j); break; } @@ -20662,35 +20662,6 @@ class binary_writer } } - /*! - @brief Writes a BSON element with key @a name and object @a value - */ - void write_bson_object_entry(const string_t& name, - const typename BasicJsonType::object_t& value) - { - write_bson_entry_header(name, 0x03); // object - write_bson_object(value); - } - - /*! - @return The size of the BSON-encoded array @a value - */ - static std::size_t calc_bson_array_size(const typename BasicJsonType::array_t& value) - { - std::size_t array_index = 0ul; - - const std::size_t embedded_document_size = std::accumulate(std::begin(value), std::end(value), static_cast(0), [&array_index](std::size_t result, const typename BasicJsonType::array_t::value_type & el) - { - // the index is built as a std::string, while calc_bson_element_size - // takes a string_t; convert explicitly, as the two are only - // implicitly convertible for some string types - const auto key = std::to_string(array_index++); - return result + calc_bson_element_size(string_t(key.data(), key.size()), el); - }); - - return sizeof(std::int32_t) + embedded_document_size + 1ul; - } - /*! @return The size of the BSON-encoded binary array @a value */ @@ -20699,29 +20670,6 @@ class binary_writer return sizeof(std::int32_t) + value.size() + 1ul; } - /*! - @brief Writes a BSON element with key @a name and array @a value - */ - void write_bson_array(const string_t& name, - const typename BasicJsonType::array_t& value) - { - write_bson_entry_header(name, 0x04); // array - write_number(to_bson_length(calc_bson_array_size(value)), true); - - std::size_t array_index = 0ul; - - for (const auto& el : value) - { - // the index is built as a std::string, while write_bson_element takes - // a string_t; convert explicitly, as the two are only implicitly - // convertible for some string types - const auto key = std::to_string(array_index++); - write_bson_element(string_t(key.data(), key.size()), el); - } - - oa.write_character(to_char_type(0x00)); - } - /*! @brief Writes a BSON element with key @a name and binary value @a value */ @@ -20743,43 +20691,37 @@ class binary_writer } /*! - @brief Calculates the size necessary to serialize the JSON value @a j with its @a name - @return The calculated size for the BSON document entry for @a j with the given @a name. + @return The size of the value of the BSON document entry for @a j, which + is neither an object nor an array */ - static std::size_t calc_bson_element_size(const string_t& name, - const BasicJsonType& j) + static std::size_t calc_bson_value_size(const BasicJsonType& j) { - const auto header_size = calc_bson_entry_header_size(name, j); switch (j.type()) { - case value_t::object: - return header_size + calc_bson_object_size(*j.m_data.m_value.object); - - case value_t::array: - return header_size + calc_bson_array_size(*j.m_data.m_value.array); - case value_t::binary: - return header_size + calc_bson_binary_size(*j.m_data.m_value.binary); + return calc_bson_binary_size(*j.m_data.m_value.binary); case value_t::boolean: - return header_size + 1ul; + return 1ul; case value_t::number_float: - return header_size + 8ul; + return 8ul; case value_t::number_integer: - return header_size + calc_bson_integer_size(j.m_data.m_value.number_integer); + return calc_bson_integer_size(j.m_data.m_value.number_integer); case value_t::number_unsigned: - return header_size + calc_bson_unsigned_size(j.m_data.m_value.number_unsigned); + return calc_bson_unsigned_size(j.m_data.m_value.number_unsigned); case value_t::string: - return header_size + calc_bson_string_size(*j.m_data.m_value.string); + return calc_bson_string_size(*j.m_data.m_value.string); case value_t::null: - return header_size + 0ul; + return 0ul; // LCOV_EXCL_START + case value_t::object: + case value_t::array: case value_t::discarded: default: JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) @@ -20789,22 +20731,13 @@ class binary_writer } /*! - @brief Serializes the JSON value @a j to BSON and associates it with the - key @a name. - @param name The name to associate with the JSON entity @a j within the - current BSON document + @brief Writes the BSON document entry with key @a name for @a j, which is + neither an object nor an array */ - void write_bson_element(const string_t& name, - const BasicJsonType& j) + void write_bson_value(const string_t& name, const BasicJsonType& j) { switch (j.type()) { - case value_t::object: - return write_bson_object_entry(name, *j.m_data.m_value.object); - - case value_t::array: - return write_bson_array(name, *j.m_data.m_value.array); - case value_t::binary: return write_bson_binary(name, *j.m_data.m_value.binary); @@ -20827,6 +20760,8 @@ class binary_writer return write_bson_null(name); // LCOV_EXCL_START + case value_t::object: + case value_t::array: case value_t::discarded: default: JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) @@ -20835,37 +20770,221 @@ class binary_writer } } - /*! - @brief Calculates the size of the BSON serialization of the given - JSON-object @a j. - @param[in] value JSON value to serialize - @pre value.type() == value_t::object - */ - static std::size_t calc_bson_object_size(const typename BasicJsonType::object_t& value) + /// @brief an object or array of the BSON document being sized or written + struct bson_frame { - const std::size_t document_size = std::accumulate(value.begin(), value.end(), static_cast(0), - [](size_t result, const typename BasicJsonType::object_t::value_type & el) + explicit bson_frame(const BasicJsonType* value_, const std::size_t size_slot_ = 0) + : value(value_) + , size_slot(size_slot_) { - return result += calc_bson_element_size(el.first, el.second); - }); + if (value->is_object()) + { + member = value->m_data.m_value.object->cbegin(); + } + } - return sizeof(std::int32_t) + document_size + 1ul; + /// the object or array + const BasicJsonType* value; + /// objects: the next member + typename BasicJsonType::object_t::const_iterator member{}; + /// arrays: the index of the next element + std::size_t index = 0; + /// @ref calc_bson_sizes only: where its size goes in the table + std::size_t size_slot; + /// @ref calc_bson_sizes only: the size of its entries seen so far + std::size_t entries_size = 0; + }; + + /*! + @brief creates the name BSON gives the array element with index @a index + @param[out] name receives the decimal index + */ + static void create_bson_index_name(const std::size_t index, string_t& name) + { + // the index is built as a std::string; convert explicitly, as the + // two are only implicitly convertible for some string types + const auto key = std::to_string(index); + name = string_t(key.data(), key.size()); } /*! - @param[in] value JSON value to serialize - @pre value.type() == value_t::object + @brief Calculates the size of every object and array in the BSON document + @a document, including the document itself. + + BSON prefixes every document and array with its size, so all of them have + to be known before the first byte is written. They are computed in a + single pass, each one from the sizes of its entries, which keeps + serializing linear in the size of the document; computing each size by + walking the entire value below it made it quadratic in the nesting depth. + The pass keeps the objects and arrays it has entered on an explicit stack, + so a deeply nested value cannot exhaust the call stack. + + @param[in] document the JSON object to serialize + @param[out] nested_sizes the sizes of the objects and arrays in + @a document, in the order they are written + @return the size of @a document + @throw out_of_range.409 if a key contains U+0000, before anything is + written */ - void write_bson_object(const typename BasicJsonType::object_t& value) + static std::size_t calc_bson_sizes(const BasicJsonType& document, std::vector& nested_sizes) { - write_number(to_bson_length(calc_bson_object_size(value)), true); + // the object or array whose entries are being sized, and the ones it + // is in; nothing is allocated unless the document nests + bson_frame current(&document); + std::vector parents; + // string_t need not be default constructible + string_t index_name("", 0); - for (const auto& el : value) + while (true) { - write_bson_element(el.first, el.second); - } + // size entries until the current object or array is done, or an + // entry is an object or array itself + const BasicJsonType* nested = nullptr; + if (current.value->is_object()) + { + const auto& object = *current.value->m_data.m_value.object; + while (nested == nullptr && current.member != object.cend()) + { + const auto& el = *current.member; + ++current.member; + current.entries_size += calc_bson_entry_header_size(el.first, el.second); + if (el.second.is_structured()) + { + nested = &el.second; + } + else + { + current.entries_size += calc_bson_value_size(el.second); + } + } + } + else + { + const auto& array = *current.value->m_data.m_value.array; + while (nested == nullptr && current.index < array.size()) + { + const BasicJsonType& el = array[current.index]; + create_bson_index_name(current.index, index_name); + current.entries_size += calc_bson_entry_header_size(index_name, el); + ++current.index; + if (el.is_structured()) + { + nested = ⪙ + } + else + { + current.entries_size += calc_bson_value_size(el); + } + } + } - oa.write_character(to_char_type(0x00)); + if (nested != nullptr) + { + // its size is added to the current one's once it is done + nested_sizes.push_back(0); + parents.push_back(std::move(current)); + current = bson_frame(nested, nested_sizes.size() - 1); + continue; + } + + // the int32 size, the entries, and the terminating null byte + const std::size_t size = sizeof(std::int32_t) + current.entries_size + 1ul; + if (parents.empty()) + { + return size; + } + nested_sizes[current.size_slot] = size; + current = std::move(parents.back()); + parents.pop_back(); + current.entries_size += size; + } + } + + /*! + @brief Serializes the JSON object @a document as a BSON document + + Writes the objects and arrays in it without the call stack, keeping the + ones it has entered on an explicit stack, so a deeply nested value + cannot exhaust the call stack. + + @param[in] document the JSON object to serialize + @pre document.type() == value_t::object + */ + void write_bson_document(const BasicJsonType& document) + { + std::vector nested_sizes; + const std::size_t document_size = calc_bson_sizes(document, nested_sizes); + write_number(to_bson_length(document_size), true); + + // the object or array whose entries are being written, and the ones + // it is in + bson_frame current(&document); + std::vector parents; + std::size_t next_size = 0; + // string_t need not be default constructible + string_t index_name("", 0); + + while (true) + { + // write entries until the current object or array is done, or an + // entry is an object or array itself + const string_t* nested_name = nullptr; + const BasicJsonType* nested = nullptr; + if (current.value->is_object()) + { + const auto& object = *current.value->m_data.m_value.object; + while (nested == nullptr && current.member != object.cend()) + { + const auto& el = *current.member; + ++current.member; + if (el.second.is_structured()) + { + nested_name = &el.first; + nested = &el.second; + } + else + { + write_bson_value(el.first, el.second); + } + } + } + else + { + const auto& array = *current.value->m_data.m_value.array; + while (nested == nullptr && current.index < array.size()) + { + const BasicJsonType& el = array[current.index]; + create_bson_index_name(current.index, index_name); + ++current.index; + if (el.is_structured()) + { + nested_name = &index_name; + nested = ⪙ + } + else + { + write_bson_value(index_name, el); + } + } + } + + if (nested != nullptr) + { + write_bson_entry_header(*nested_name, nested->is_object() ? 0x03 : 0x04); + write_number(to_bson_length(nested_sizes[next_size++]), true); + parents.push_back(std::move(current)); + current = bson_frame(nested); + continue; + } + + oa.write_character(to_char_type(0x00)); + if (parents.empty()) + { + return; + } + current = std::move(parents.back()); + parents.pop_back(); + } } ////////// diff --git a/tests/src/unit-bson.cpp b/tests/src/unit-bson.cpp index 4808c3166..03cde5376 100644 --- a/tests/src/unit-bson.cpp +++ b/tests/src/unit-bson.cpp @@ -49,7 +49,7 @@ using huge_binary_json = nlohmann::basic_json < // for *object keys* (e.g. "s" or "nested" below). Only the designated test // value is meant to lie about its size - if every huge_string_t (including // keys) reported a huge size, the running totals computed while walking the -// BSON document (see calc_bson_object_size & friends in binary_writer.hpp) +// BSON document (see calc_bson_sizes in binary_writer.hpp) // would need more than 32 bits, and on platforms where std::size_t is only // 32 bits wide that arithmetic would silently wrap around, producing wrong // (or even unguarded) lengths. The fake size is therefore opt-in via @@ -1698,6 +1698,70 @@ TEST_CASE("BSON roundtrips" * doctest::skip()) } } +TEST_CASE("BSON: deeply nested values") +{ + SECTION("documents and arrays round-trip at every depth") + { + // nested documents and arrays, with siblings on every level, so + // every length prefix covers entries of both kinds + json value = "leaf"; + for (std::size_t depth = 0; depth <= 300; ++depth) + { + CAPTURE(depth); + const json document = {{"value", value}, {"n", depth}}; + CHECK(json::from_bson(json::to_bson(document)) == document); + +value = depth % 2 == 0 ? json{{"a", std::move(value)}, {"b", {1, "x"}}} : + json::array({std::move(value), depth, json::object()}); + } + } + + SECTION("a key containing U+0000 is rejected before anything is written") + { + json value = json::object({{std::string("bad\0key", 7), 1}}); + for (std::size_t depth = 0; depth < 200; ++depth) + { + value = json{{"a", {{"b", 1}}}, {"z", std::move(value)}}; + } + std::vector output; + CHECK_THROWS_AS(json::to_bson(value, output), json::out_of_range&); + CHECK(output.empty()); + } + + SECTION("values nested too deeply for the call stack (#5392)") + { + // serializing recursed once per nesting level, and computed every + // nested document's length by walking everything below it again. + // The values are only parsed, serialized and walked, never copied or + // compared, since those recurse too. + const std::size_t depth = 100000; + for (const bool objects : + { + false, true + }) + { + CAPTURE(objects); + std::string text = "{\"a\":"; + for (std::size_t i = 0; i < depth; ++i) + { + text += objects ? "{\"a\":" : "["; + } + text += "1"; + text.append(depth, objects ? '}' : ']'); + text += "}"; + + const auto bson = json::to_bson(json::parse(text)); + const auto result = json::from_bson(bson); + const json* p = &result.at("a"); + for (std::size_t i = 0; i < depth; ++i) + { + p = objects ? &p->at("a") : &p->at(0); + } + CHECK(*p == 1); + } + } +} + TEST_CASE("Invalid document size handling") { SECTION("document size must be at least 5") From f422b753cca0143b2c8545df2e1422171a2f9cd5 Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Fri, 25 Sep 2026 21:57:28 +0200 Subject: [PATCH 54/64] Bump the codeql-action group across 1 directory with 4 updates (#5576) Bumps the codeql-action group with 4 updates in the / directory: [github/codeql-action/init](https://github.com/github/codeql-action), [github/codeql-action/autobuild](https://github.com/github/codeql-action), [github/codeql-action/analyze](https://github.com/github/codeql-action) and [github/codeql-action/upload-sarif](https://github.com/github/codeql-action). Updates `github/codeql-action/init` from 4.38.0 to 4.38.1 - [Release notes](https://github.com/github/codeql-action/releases) - [Changelog](https://github.com/github/codeql-action/blob/main/CHANGELOG.md) - [Commits](https://github.com/github/codeql-action/compare/b96794f015dfd88f77b49b1c93e0fa7110f94c63...1c5b675653bb5c22dbe9b12b556ec555138e09fd) Updates `github/codeql-action/autobuild` from 4.38.0 to 4.38.1 - [Release notes](https://github.com/github/codeql-action/releases) - [Changelog](https://github.com/github/codeql-action/blob/main/CHANGELOG.md) - [Commits](https://github.com/github/codeql-action/compare/b96794f015dfd88f77b49b1c93e0fa7110f94c63...1c5b675653bb5c22dbe9b12b556ec555138e09fd) Updates `github/codeql-action/analyze` from 4.38.0 to 4.38.1 - [Release notes](https://github.com/github/codeql-action/releases) - [Changelog](https://github.com/github/codeql-action/blob/main/CHANGELOG.md) - [Commits](https://github.com/github/codeql-action/compare/b96794f015dfd88f77b49b1c93e0fa7110f94c63...1c5b675653bb5c22dbe9b12b556ec555138e09fd) Updates `github/codeql-action/upload-sarif` from 4.38.0 to 4.38.1 - [Release notes](https://github.com/github/codeql-action/releases) - [Changelog](https://github.com/github/codeql-action/blob/main/CHANGELOG.md) - [Commits](https://github.com/github/codeql-action/compare/b96794f015dfd88f77b49b1c93e0fa7110f94c63...1c5b675653bb5c22dbe9b12b556ec555138e09fd) --- updated-dependencies: - dependency-name: github/codeql-action/analyze dependency-version: 4.38.1 dependency-type: direct:production update-type: version-update:semver-patch dependency-group: codeql-action - dependency-name: github/codeql-action/autobuild dependency-version: 4.38.1 dependency-type: direct:production update-type: version-update:semver-patch dependency-group: codeql-action - dependency-name: github/codeql-action/init dependency-version: 4.38.1 dependency-type: direct:production update-type: version-update:semver-patch dependency-group: codeql-action - dependency-name: github/codeql-action/upload-sarif dependency-version: 4.38.1 dependency-type: direct:production update-type: version-update:semver-patch dependency-group: codeql-action ... Signed-off-by: dependabot[bot] Co-authored-by: dependabot[bot] <49699333+dependabot[bot]@users.noreply.github.com> --- .github/workflows/codeql-analysis.yml | 6 +++--- .github/workflows/flawfinder.yml | 2 +- .github/workflows/scorecards.yml | 2 +- .github/workflows/semgrep.yml | 2 +- 4 files changed, 6 insertions(+), 6 deletions(-) diff --git a/.github/workflows/codeql-analysis.yml b/.github/workflows/codeql-analysis.yml index 1c2e73cd1..74ecdfa91 100644 --- a/.github/workflows/codeql-analysis.yml +++ b/.github/workflows/codeql-analysis.yml @@ -38,14 +38,14 @@ jobs: # Initializes the CodeQL tools for scanning. - name: Initialize CodeQL - uses: github/codeql-action/init@b96794f015dfd88f77b49b1c93e0fa7110f94c63 # v4.38.0 + uses: github/codeql-action/init@1c5b675653bb5c22dbe9b12b556ec555138e09fd # v4.38.1 with: languages: c-cpp # Autobuild attempts to build any compiled languages (C/C++, C#, or Java). # If this step fails, then you should remove it and run the build manually (see below) - name: Autobuild - uses: github/codeql-action/autobuild@b96794f015dfd88f77b49b1c93e0fa7110f94c63 # v4.38.0 + uses: github/codeql-action/autobuild@1c5b675653bb5c22dbe9b12b556ec555138e09fd # v4.38.1 - name: Perform CodeQL Analysis - uses: github/codeql-action/analyze@b96794f015dfd88f77b49b1c93e0fa7110f94c63 # v4.38.0 + uses: github/codeql-action/analyze@1c5b675653bb5c22dbe9b12b556ec555138e09fd # v4.38.1 diff --git a/.github/workflows/flawfinder.yml b/.github/workflows/flawfinder.yml index 753b81e0d..9f3b5562a 100644 --- a/.github/workflows/flawfinder.yml +++ b/.github/workflows/flawfinder.yml @@ -47,6 +47,6 @@ jobs: output: 'flawfinder_results.sarif' - name: Upload analysis results to GitHub Security tab - uses: github/codeql-action/upload-sarif@b96794f015dfd88f77b49b1c93e0fa7110f94c63 # v4.38.0 + uses: github/codeql-action/upload-sarif@1c5b675653bb5c22dbe9b12b556ec555138e09fd # v4.38.1 with: sarif_file: ${{github.workspace}}/flawfinder_results.sarif diff --git a/.github/workflows/scorecards.yml b/.github/workflows/scorecards.yml index 96f87526a..5e4b0c315 100644 --- a/.github/workflows/scorecards.yml +++ b/.github/workflows/scorecards.yml @@ -80,6 +80,6 @@ jobs: # Upload the results to GitHub's code scanning dashboard. - name: "Upload to code-scanning" - uses: github/codeql-action/upload-sarif@b96794f015dfd88f77b49b1c93e0fa7110f94c63 # v4.38.0 + uses: github/codeql-action/upload-sarif@1c5b675653bb5c22dbe9b12b556ec555138e09fd # v4.38.1 with: sarif_file: results.sarif diff --git a/.github/workflows/semgrep.yml b/.github/workflows/semgrep.yml index 922c9ca86..434e1f3bb 100644 --- a/.github/workflows/semgrep.yml +++ b/.github/workflows/semgrep.yml @@ -65,7 +65,7 @@ jobs: # Upload SARIF file generated in previous step - name: Upload SARIF file - uses: github/codeql-action/upload-sarif@b96794f015dfd88f77b49b1c93e0fa7110f94c63 # v4.38.0 + uses: github/codeql-action/upload-sarif@1c5b675653bb5c22dbe9b12b556ec555138e09fd # v4.38.1 with: sarif_file: semgrep.sarif if: always() From fe4a544c7eb9ca44b63d683cf8e6593380f4c1ad Mon Sep 17 00:00:00 2001 From: Dadi Reddy Sai Praneeth Reddy Date: Sun, 27 Sep 2026 17:47:32 +0530 Subject: [PATCH 55/64] handled when size exceed uint32 (#5515) * handled when size exceed uint32 Signed-off-by: dsp0redy * addressed review comments Signed-off-by: dsp0redy * updated unit test Signed-off-by: dsp0redy * added amalgamation patch Signed-off-by: dsp0redy --------- Signed-off-by: dsp0redy --- .../nlohmann/detail/output/binary_writer.hpp | 20 +++ single_include/nlohmann/json.hpp | 20 +++ tests/src/unit-msgpack.cpp | 168 ++++++++++++++++++ 3 files changed, 208 insertions(+) diff --git a/include/nlohmann/detail/output/binary_writer.hpp b/include/nlohmann/detail/output/binary_writer.hpp index 5a0722a18..42ee60389 100644 --- a/include/nlohmann/detail/output/binary_writer.hpp +++ b/include/nlohmann/detail/output/binary_writer.hpp @@ -630,6 +630,11 @@ class binary_writer oa.write_character(to_char_type(0xDB)); write_number(static_cast(N)); } + else + { + JSON_THROW(out_of_range::create(412, concat("MessagePack size ", std::to_string(N), " exceeds maximum of ", + std::to_string((std::numeric_limits::max)())), &j)); + } // step 2: write the string oa.write_characters( @@ -659,6 +664,11 @@ class binary_writer oa.write_character(to_char_type(0xDD)); write_number(static_cast(N)); } + else + { + JSON_THROW(out_of_range::create(412, concat("MessagePack size ", std::to_string(N), " exceeds maximum of ", + std::to_string((std::numeric_limits::max)())), &j)); + } // step 2: write each element for (const auto& el : *j.m_data.m_value.array) @@ -736,6 +746,11 @@ class binary_writer oa.write_character(to_char_type(output_type)); write_number(static_cast(N)); } + else + { + JSON_THROW(out_of_range::create(412, concat("MessagePack size ", std::to_string(N), " exceeds maximum of ", + std::to_string((std::numeric_limits::max)())), &j)); + } // step 1.5: if this is an ext type, write the subtype if (use_ext) @@ -777,6 +792,11 @@ class binary_writer oa.write_character(to_char_type(0xDF)); write_number(static_cast(N)); } + else + { + JSON_THROW(out_of_range::create(412, concat("MessagePack size ", std::to_string(N), " exceeds maximum of ", + std::to_string((std::numeric_limits::max)())), &j)); + } // step 2: write each element for (const auto& el : *j.m_data.m_value.object) diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 4981c3b02..814b8f411 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -20095,6 +20095,11 @@ class binary_writer oa.write_character(to_char_type(0xDB)); write_number(static_cast(N)); } + else + { + JSON_THROW(out_of_range::create(412, concat("MessagePack size ", std::to_string(N), " exceeds maximum of ", + std::to_string((std::numeric_limits::max)())), &j)); + } // step 2: write the string oa.write_characters( @@ -20124,6 +20129,11 @@ class binary_writer oa.write_character(to_char_type(0xDD)); write_number(static_cast(N)); } + else + { + JSON_THROW(out_of_range::create(412, concat("MessagePack size ", std::to_string(N), " exceeds maximum of ", + std::to_string((std::numeric_limits::max)())), &j)); + } // step 2: write each element for (const auto& el : *j.m_data.m_value.array) @@ -20201,6 +20211,11 @@ class binary_writer oa.write_character(to_char_type(output_type)); write_number(static_cast(N)); } + else + { + JSON_THROW(out_of_range::create(412, concat("MessagePack size ", std::to_string(N), " exceeds maximum of ", + std::to_string((std::numeric_limits::max)())), &j)); + } // step 1.5: if this is an ext type, write the subtype if (use_ext) @@ -20242,6 +20257,11 @@ class binary_writer oa.write_character(to_char_type(0xDF)); write_number(static_cast(N)); } + else + { + JSON_THROW(out_of_range::create(412, concat("MessagePack size ", std::to_string(N), " exceeds maximum of ", + std::to_string((std::numeric_limits::max)())), &j)); + } // step 2: write each element for (const auto& el : *j.m_data.m_value.object) diff --git a/tests/src/unit-msgpack.cpp b/tests/src/unit-msgpack.cpp index 805b03a13..8168b88ae 100644 --- a/tests/src/unit-msgpack.cpp +++ b/tests/src/unit-msgpack.cpp @@ -2150,3 +2150,171 @@ TEST_CASE("MessagePack with std::byte") } } #endif + +template> +struct huge_array : std::vector +{ + using base = std::vector; + using base::base; + + bool fake_size = false; + + std::size_t size() const noexcept + { + if (fake_size) + { + return (std::numeric_limits::max)() + 1ULL; + } + + return base::size(); + } +}; + +using huge_array_json = nlohmann::basic_json < + std::map, huge_array, std::string, bool, std::int64_t, std::uint64_t, + double, std::allocator, nlohmann::adl_serializer, + std::vector, void >; + +TEST_CASE("MessagePack Size above uint32 for array") +{ + huge_array_json j = huge_array_json::array(); + + j.push_back(1); + j.push_back(2); + j.push_back(3); + + auto& array = j.get_ref(); + array.fake_size = true; + + CHECK_THROWS_WITH_AS( + huge_array_json::to_msgpack(j), + "[json.exception.out_of_range.412] MessagePack size 4294967296 exceeds maximum of 4294967295", + json::out_of_range&); + + array.fake_size = false; +} + +template, + typename A = std::allocator>> + struct huge_map : std::map +{ + using base = std::map; + using base::base; + + bool fake_size = false; + + std::size_t size() const noexcept + { + if (fake_size) + { + return static_cast(UINT32_MAX) + 1ULL; + } + + return base::size(); + } +}; + +using huge_object_json = nlohmann::basic_json < + huge_map, + std::vector, + std::string, + bool, + std::int64_t, + std::uint64_t, + double, + std::allocator, + nlohmann::adl_serializer, + std::vector, + void >; + +TEST_CASE("MessagePack Size above uint32 for object") +{ + + huge_object_json j = huge_object_json::object(); + + j["one"] = 1; + j["two"] = 2; + + auto& object = j.get_ref(); + object.fake_size = true; + + CHECK_THROWS_WITH_AS( + huge_object_json::to_msgpack(j), + "[json.exception.out_of_range.412] MessagePack size 4294967296 exceeds maximum of 4294967295", + json::out_of_range&); + + object.fake_size = false; +} + +struct huge_string : std::string +{ + using std::string::string; + + std::size_t size() const noexcept + { + return static_cast(UINT32_MAX) + 1ULL; + } +}; + +using huge_string_json = nlohmann::basic_json < + std::map, + std::vector, + huge_string, + bool, + std::int64_t, + std::uint64_t, + double, + std::allocator, + nlohmann::adl_serializer, + std::vector, + void >; + +TEST_CASE("MessagePack Size above uint32 for string") +{ + + huge_string_json j = "hello"; + + CHECK_THROWS_WITH_AS( + huge_string_json::to_msgpack(j), + "[json.exception.out_of_range.412] MessagePack size 4294967296 exceeds maximum of 4294967295", + json::out_of_range&); +} + +struct huge_binary : std::vector +{ + using std::vector::vector; + + std::size_t size() const noexcept + { + return static_cast(UINT32_MAX) + 1ULL; + } +}; + +using huge_binary_json = nlohmann::basic_json < + std::map, + std::vector, + std::string, + bool, + std::int64_t, + std::uint64_t, + double, + std::allocator, + nlohmann::adl_serializer, + huge_binary, + void >; + +TEST_CASE("MessagePack Size above uint32 for binary") +{ + + huge_binary_json j = huge_binary_json::binary(huge_binary{}); + + j.get_binary().push_back(0x01); + j.get_binary().push_back(0x02); + + CHECK_THROWS_WITH_AS( + huge_binary_json::to_msgpack(j), + "[json.exception.out_of_range.412] MessagePack size 4294967296 exceeds maximum of 4294967295", + json::out_of_range&); +} + From 4fa95d98105ac30c02293675e75d0651fea2d60a Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Sun, 27 Sep 2026 14:18:46 +0200 Subject: [PATCH 56/64] Remove unreachable branches from the binary writer (#5583) Coverage reported conditions in the binary writer that can never be false, and marked the code behind them with LCOV_EXCL. Remove them instead of excluding them: - CBOR writes the length of a string, binary value, array, or object exactly like an unsigned integer, only with another major type. One function, write_cbor_head(), now writes both, so the integer tests cover every width and the four excluded 64-bit length branches are gone. - A last `else if` whose condition holds for every remaining value (an unsigned value at most UINT64_MAX, a signed one in the range of int64_t) is now a plain `else`. - Whether a signed integer fits into an int64 for UBJSON and BJData is decided by its type at compile time. Only an integer type wider than 64 bits gets a range check and the high-precision fallback. - The private get_impl(boolean_t*) was never called. The UBJSON type prefix 'H' of an optimized container of unsigned integers beyond the range of int64 was reachable although excluded; it is tested now. The output is unchanged. Signed-off-by: Niels Lohmann --- .../nlohmann/detail/output/binary_writer.hpp | 329 ++++++----------- include/nlohmann/json.hpp | 11 - single_include/nlohmann/json.hpp | 340 ++++++------------ tests/src/unit-ubjson.cpp | 15 + 4 files changed, 235 insertions(+), 460 deletions(-) diff --git a/include/nlohmann/detail/output/binary_writer.hpp b/include/nlohmann/detail/output/binary_writer.hpp index 42ee60389..9c41e2962 100644 --- a/include/nlohmann/detail/output/binary_writer.hpp +++ b/include/nlohmann/detail/output/binary_writer.hpp @@ -168,92 +168,20 @@ class binary_writer if (j.m_data.m_value.number_integer >= 0) { // CBOR does not differentiate between positive signed - // integers and unsigned integers. Therefore, we used the - // code from the value_t::number_unsigned case here. - if (j.m_data.m_value.number_integer <= 0x17) - { - write_number(static_cast(j.m_data.m_value.number_integer)); - } - else if (j.m_data.m_value.number_integer <= (std::numeric_limits::max)()) - { - oa.write_character(to_char_type(0x18)); - write_number(static_cast(j.m_data.m_value.number_integer)); - } - else if (j.m_data.m_value.number_integer <= (std::numeric_limits::max)()) - { - oa.write_character(to_char_type(0x19)); - write_number(static_cast(j.m_data.m_value.number_integer)); - } - else if (j.m_data.m_value.number_integer <= (std::numeric_limits::max)()) - { - oa.write_character(to_char_type(0x1A)); - write_number(static_cast(j.m_data.m_value.number_integer)); - } - else - { - oa.write_character(to_char_type(0x1B)); - write_number(static_cast(j.m_data.m_value.number_integer)); - } + // integers and unsigned integers + write_cbor_head(0x00, static_cast(j.m_data.m_value.number_integer)); } else { - // The conversions below encode the sign in the first - // byte, and the value is converted to a positive number. - const auto positive_number = -1 - j.m_data.m_value.number_integer; - if (j.m_data.m_value.number_integer >= -24) - { - write_number(static_cast(0x20 + positive_number)); - } - else if (positive_number <= (std::numeric_limits::max)()) - { - oa.write_character(to_char_type(0x38)); - write_number(static_cast(positive_number)); - } - else if (positive_number <= (std::numeric_limits::max)()) - { - oa.write_character(to_char_type(0x39)); - write_number(static_cast(positive_number)); - } - else if (positive_number <= (std::numeric_limits::max)()) - { - oa.write_character(to_char_type(0x3A)); - write_number(static_cast(positive_number)); - } - else - { - oa.write_character(to_char_type(0x3B)); - write_number(static_cast(positive_number)); - } + // a negative integer n is encoded as -1 - n + write_cbor_head(0x20, static_cast(-1 - j.m_data.m_value.number_integer)); } break; } case value_t::number_unsigned: { - if (j.m_data.m_value.number_unsigned <= 0x17) - { - write_number(static_cast(j.m_data.m_value.number_unsigned)); - } - else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) - { - oa.write_character(to_char_type(0x18)); - write_number(static_cast(j.m_data.m_value.number_unsigned)); - } - else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) - { - oa.write_character(to_char_type(0x19)); - write_number(static_cast(j.m_data.m_value.number_unsigned)); - } - else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) - { - oa.write_character(to_char_type(0x1A)); - write_number(static_cast(j.m_data.m_value.number_unsigned)); - } - else - { - oa.write_character(to_char_type(0x1B)); - write_number(static_cast(j.m_data.m_value.number_unsigned)); - } + write_cbor_head(0x00, j.m_data.m_value.number_unsigned); break; } @@ -283,33 +211,7 @@ class binary_writer case value_t::string: { // step 1: write control byte and the string length - const auto N = j.m_data.m_value.string->size(); - if (N <= 0x17) - { - write_number(static_cast(0x60 + N)); - } - else if (N <= (std::numeric_limits::max)()) - { - oa.write_character(to_char_type(0x78)); - write_number(static_cast(N)); - } - else if (N <= (std::numeric_limits::max)()) - { - oa.write_character(to_char_type(0x79)); - write_number(static_cast(N)); - } - else if (N <= (std::numeric_limits::max)()) - { - oa.write_character(to_char_type(0x7A)); - write_number(static_cast(N)); - } - // LCOV_EXCL_START - else if (N <= (std::numeric_limits::max)()) - { - oa.write_character(to_char_type(0x7B)); - write_number(static_cast(N)); - } - // LCOV_EXCL_STOP + write_cbor_head(0x60, j.m_data.m_value.string->size()); // step 2: write the string oa.write_characters( @@ -321,33 +223,7 @@ class binary_writer case value_t::array: { // step 1: write control byte and the array size - const auto N = j.m_data.m_value.array->size(); - if (N <= 0x17) - { - write_number(static_cast(0x80 + N)); - } - else if (N <= (std::numeric_limits::max)()) - { - oa.write_character(to_char_type(0x98)); - write_number(static_cast(N)); - } - else if (N <= (std::numeric_limits::max)()) - { - oa.write_character(to_char_type(0x99)); - write_number(static_cast(N)); - } - else if (N <= (std::numeric_limits::max)()) - { - oa.write_character(to_char_type(0x9A)); - write_number(static_cast(N)); - } - // LCOV_EXCL_START - else if (N <= (std::numeric_limits::max)()) - { - oa.write_character(to_char_type(0x9B)); - write_number(static_cast(N)); - } - // LCOV_EXCL_STOP + write_cbor_head(0x80, j.m_data.m_value.array->size()); // step 2: write each element for (const auto& el : *j.m_data.m_value.array) @@ -376,7 +252,7 @@ class binary_writer write_number(static_cast(0xda)); write_number(static_cast(j.m_data.m_value.binary->subtype())); } - else if (j.m_data.m_value.binary->subtype() <= (std::numeric_limits::max)()) + else { write_number(static_cast(0xdb)); write_number(static_cast(j.m_data.m_value.binary->subtype())); @@ -385,32 +261,7 @@ class binary_writer // step 1: write control byte and the binary array size const auto N = j.m_data.m_value.binary->size(); - if (N <= 0x17) - { - write_number(static_cast(0x40 + N)); - } - else if (N <= (std::numeric_limits::max)()) - { - oa.write_character(to_char_type(0x58)); - write_number(static_cast(N)); - } - else if (N <= (std::numeric_limits::max)()) - { - oa.write_character(to_char_type(0x59)); - write_number(static_cast(N)); - } - else if (N <= (std::numeric_limits::max)()) - { - oa.write_character(to_char_type(0x5A)); - write_number(static_cast(N)); - } - // LCOV_EXCL_START - else if (N <= (std::numeric_limits::max)()) - { - oa.write_character(to_char_type(0x5B)); - write_number(static_cast(N)); - } - // LCOV_EXCL_STOP + write_cbor_head(0x40, N); // step 2: write each element oa.write_characters( @@ -423,33 +274,7 @@ class binary_writer case value_t::object: { // step 1: write control byte and the object size - const auto N = j.m_data.m_value.object->size(); - if (N <= 0x17) - { - write_number(static_cast(0xA0 + N)); - } - else if (N <= (std::numeric_limits::max)()) - { - oa.write_character(to_char_type(0xB8)); - write_number(static_cast(N)); - } - else if (N <= (std::numeric_limits::max)()) - { - oa.write_character(to_char_type(0xB9)); - write_number(static_cast(N)); - } - else if (N <= (std::numeric_limits::max)()) - { - oa.write_character(to_char_type(0xBA)); - write_number(static_cast(N)); - } - // LCOV_EXCL_START - else if (N <= (std::numeric_limits::max)()) - { - oa.write_character(to_char_type(0xBB)); - write_number(static_cast(N)); - } - // LCOV_EXCL_STOP + write_cbor_head(0xA0, j.m_data.m_value.object->size()); // step 2: write each element for (const auto& el : *j.m_data.m_value.object) @@ -517,7 +342,7 @@ class binary_writer oa.write_character(to_char_type(0xCE)); write_number(static_cast(j.m_data.m_value.number_integer)); } - else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) + else { // uint 64 oa.write_character(to_char_type(0xCF)); @@ -552,8 +377,7 @@ class binary_writer oa.write_character(to_char_type(0xD2)); write_number(static_cast(j.m_data.m_value.number_integer)); } - else if (j.m_data.m_value.number_integer >= (std::numeric_limits::min)() && - j.m_data.m_value.number_integer <= (std::numeric_limits::max)()) + else { // int 64 oa.write_character(to_char_type(0xD3)); @@ -588,7 +412,7 @@ class binary_writer oa.write_character(to_char_type(0xCE)); write_number(static_cast(j.m_data.m_value.number_integer)); } - else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) + else { // uint 64 oa.write_character(to_char_type(0xCF)); @@ -1546,6 +1370,46 @@ class binary_writer // CBOR // ////////// + /*! + @brief write the head of a CBOR data item + + The head is the major type in the upper three bits of the first byte and + an argument - an unsigned integer, the length of a string, the number of + elements of a container - in the shortest of its encodings: in the lower + five bits of the first byte itself if it is at most 23, otherwise in the + 1, 2, 4, or 8 bytes that follow (RFC 8949, section 3). + + @param[in] major_type the major type, shifted into the upper three bits + @param[in] argument the argument of the data item + */ + void write_cbor_head(const std::uint8_t major_type, const std::uint64_t argument) + { + if (argument <= 0x17) + { + write_number(static_cast(major_type + argument)); + } + else if (argument <= (std::numeric_limits::max)()) + { + oa.write_character(to_char_type(static_cast(major_type + 0x18))); + write_number(static_cast(argument)); + } + else if (argument <= (std::numeric_limits::max)()) + { + oa.write_character(to_char_type(static_cast(major_type + 0x19))); + write_number(static_cast(argument)); + } + else if (argument <= (std::numeric_limits::max)()) + { + oa.write_character(to_char_type(static_cast(major_type + 0x1A))); + write_number(static_cast(argument)); + } + else + { + oa.write_character(to_char_type(static_cast(major_type + 0x1B))); + write_number(argument); + } + } + static constexpr CharType get_cbor_float_prefix(float /*unused*/) { return to_char_type(0xFA); // Single-Precision Float @@ -1651,7 +1515,7 @@ class binary_writer } write_number(static_cast(n), use_bjdata); } - else if (use_bjdata && n <= (std::numeric_limits::max)()) + else if (use_bjdata) { if (add_prefix) { @@ -1731,30 +1595,59 @@ class binary_writer } write_number(static_cast(n), use_bjdata); } - else if ((std::numeric_limits::min)() <= n && n <= (std::numeric_limits::max)()) - { - if (add_prefix) - { - oa.write_character(to_char_type('L')); // int64 - } - write_number(static_cast(n), use_bjdata); - } - // LCOV_EXCL_START else { - if (add_prefix) - { - oa.write_character(to_char_type('H')); // high-precision number - } - - const auto number = BasicJsonType(n).dump(); - write_number_with_ubjson_prefix(number.size(), true, use_bjdata); - for (std::size_t i = 0; i < number.size(); ++i) - { - oa.write_character(to_char_type(static_cast(number[i]))); - } + // every value of an integer type of at most 64 bits fits into an + // int64; only a wider type needs a range check + write_ubjson_int64_or_high_precision(n, add_prefix, use_bjdata, + std::integral_constant < bool, std::numeric_limits::digits <= std::numeric_limits::digits > {}); } - // LCOV_EXCL_STOP + } + + template + void write_ubjson_int64_or_high_precision(const NumberType n, const bool add_prefix, const bool use_bjdata, std::true_type /*fits_int64*/) + { + if (add_prefix) + { + oa.write_character(to_char_type('L')); // int64 + } + write_number(static_cast(n), use_bjdata); + } + + template + void write_ubjson_int64_or_high_precision(const NumberType n, const bool add_prefix, const bool use_bjdata, std::false_type /*fits_int64*/) + { + if ((std::numeric_limits::min)() <= n && n <= (std::numeric_limits::max)()) + { + write_ubjson_int64_or_high_precision(n, add_prefix, use_bjdata, std::true_type {}); + return; + } + + if (add_prefix) + { + oa.write_character(to_char_type('H')); // high-precision number + } + + const auto number = BasicJsonType(n).dump(); + write_number_with_ubjson_prefix(number.size(), true, use_bjdata); + for (std::size_t i = 0; i < number.size(); ++i) + { + oa.write_character(to_char_type(static_cast(number[i]))); + } + } + + template + static constexpr CharType ubjson_int64_or_high_precision_prefix(const NumberType /*n*/, std::true_type /*fits_int64*/) noexcept + { + return 'L'; + } + + template + static CharType ubjson_int64_or_high_precision_prefix(const NumberType n, std::false_type /*fits_int64*/) noexcept + { + // anything outside of the range of an int64 is treated as a + // high-precision number + return ((std::numeric_limits::min)() <= n && n <= (std::numeric_limits::max)()) ? 'L' : 'H'; } /*! @@ -1796,12 +1689,10 @@ class binary_writer { return 'm'; } - if ((std::numeric_limits::min)() <= j.m_data.m_value.number_integer && j.m_data.m_value.number_integer <= (std::numeric_limits::max)()) - { - return 'L'; - } - // anything else is treated as a high-precision number - return 'H'; // LCOV_EXCL_LINE + // every value of an integer type of at most 64 bits fits into + // an int64; only a wider type needs a range check + return ubjson_int64_or_high_precision_prefix(j.m_data.m_value.number_integer, + std::integral_constant < bool, std::numeric_limits::digits <= std::numeric_limits::digits > {}); } case value_t::number_unsigned: @@ -1834,12 +1725,12 @@ class binary_writer { return 'L'; } - if (use_bjdata && j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) + if (use_bjdata) { return 'M'; } // anything else is treated as a high-precision number - return 'H'; // LCOV_EXCL_LINE + return 'H'; } case value_t::number_float: diff --git a/include/nlohmann/json.hpp b/include/nlohmann/json.hpp index c578fd11a..d18baf599 100644 --- a/include/nlohmann/json.hpp +++ b/include/nlohmann/json.hpp @@ -2157,17 +2157,6 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // value access // ////////////////// - /// get a boolean (explicit) - boolean_t get_impl(boolean_t* /*unused*/) const - { - if (JSON_HEDLEY_LIKELY(is_boolean())) - { - return m_data.m_value.boolean; - } - - JSON_THROW(type_error::create(302, detail::concat("type must be boolean, but is ", type_name()), this)); - } - /// get a pointer to the value (object) object_t* get_impl_ptr(object_t* /*unused*/) noexcept { diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 814b8f411..b5938be0e 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -19633,92 +19633,20 @@ class binary_writer if (j.m_data.m_value.number_integer >= 0) { // CBOR does not differentiate between positive signed - // integers and unsigned integers. Therefore, we used the - // code from the value_t::number_unsigned case here. - if (j.m_data.m_value.number_integer <= 0x17) - { - write_number(static_cast(j.m_data.m_value.number_integer)); - } - else if (j.m_data.m_value.number_integer <= (std::numeric_limits::max)()) - { - oa.write_character(to_char_type(0x18)); - write_number(static_cast(j.m_data.m_value.number_integer)); - } - else if (j.m_data.m_value.number_integer <= (std::numeric_limits::max)()) - { - oa.write_character(to_char_type(0x19)); - write_number(static_cast(j.m_data.m_value.number_integer)); - } - else if (j.m_data.m_value.number_integer <= (std::numeric_limits::max)()) - { - oa.write_character(to_char_type(0x1A)); - write_number(static_cast(j.m_data.m_value.number_integer)); - } - else - { - oa.write_character(to_char_type(0x1B)); - write_number(static_cast(j.m_data.m_value.number_integer)); - } + // integers and unsigned integers + write_cbor_head(0x00, static_cast(j.m_data.m_value.number_integer)); } else { - // The conversions below encode the sign in the first - // byte, and the value is converted to a positive number. - const auto positive_number = -1 - j.m_data.m_value.number_integer; - if (j.m_data.m_value.number_integer >= -24) - { - write_number(static_cast(0x20 + positive_number)); - } - else if (positive_number <= (std::numeric_limits::max)()) - { - oa.write_character(to_char_type(0x38)); - write_number(static_cast(positive_number)); - } - else if (positive_number <= (std::numeric_limits::max)()) - { - oa.write_character(to_char_type(0x39)); - write_number(static_cast(positive_number)); - } - else if (positive_number <= (std::numeric_limits::max)()) - { - oa.write_character(to_char_type(0x3A)); - write_number(static_cast(positive_number)); - } - else - { - oa.write_character(to_char_type(0x3B)); - write_number(static_cast(positive_number)); - } + // a negative integer n is encoded as -1 - n + write_cbor_head(0x20, static_cast(-1 - j.m_data.m_value.number_integer)); } break; } case value_t::number_unsigned: { - if (j.m_data.m_value.number_unsigned <= 0x17) - { - write_number(static_cast(j.m_data.m_value.number_unsigned)); - } - else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) - { - oa.write_character(to_char_type(0x18)); - write_number(static_cast(j.m_data.m_value.number_unsigned)); - } - else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) - { - oa.write_character(to_char_type(0x19)); - write_number(static_cast(j.m_data.m_value.number_unsigned)); - } - else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) - { - oa.write_character(to_char_type(0x1A)); - write_number(static_cast(j.m_data.m_value.number_unsigned)); - } - else - { - oa.write_character(to_char_type(0x1B)); - write_number(static_cast(j.m_data.m_value.number_unsigned)); - } + write_cbor_head(0x00, j.m_data.m_value.number_unsigned); break; } @@ -19748,33 +19676,7 @@ class binary_writer case value_t::string: { // step 1: write control byte and the string length - const auto N = j.m_data.m_value.string->size(); - if (N <= 0x17) - { - write_number(static_cast(0x60 + N)); - } - else if (N <= (std::numeric_limits::max)()) - { - oa.write_character(to_char_type(0x78)); - write_number(static_cast(N)); - } - else if (N <= (std::numeric_limits::max)()) - { - oa.write_character(to_char_type(0x79)); - write_number(static_cast(N)); - } - else if (N <= (std::numeric_limits::max)()) - { - oa.write_character(to_char_type(0x7A)); - write_number(static_cast(N)); - } - // LCOV_EXCL_START - else if (N <= (std::numeric_limits::max)()) - { - oa.write_character(to_char_type(0x7B)); - write_number(static_cast(N)); - } - // LCOV_EXCL_STOP + write_cbor_head(0x60, j.m_data.m_value.string->size()); // step 2: write the string oa.write_characters( @@ -19786,33 +19688,7 @@ class binary_writer case value_t::array: { // step 1: write control byte and the array size - const auto N = j.m_data.m_value.array->size(); - if (N <= 0x17) - { - write_number(static_cast(0x80 + N)); - } - else if (N <= (std::numeric_limits::max)()) - { - oa.write_character(to_char_type(0x98)); - write_number(static_cast(N)); - } - else if (N <= (std::numeric_limits::max)()) - { - oa.write_character(to_char_type(0x99)); - write_number(static_cast(N)); - } - else if (N <= (std::numeric_limits::max)()) - { - oa.write_character(to_char_type(0x9A)); - write_number(static_cast(N)); - } - // LCOV_EXCL_START - else if (N <= (std::numeric_limits::max)()) - { - oa.write_character(to_char_type(0x9B)); - write_number(static_cast(N)); - } - // LCOV_EXCL_STOP + write_cbor_head(0x80, j.m_data.m_value.array->size()); // step 2: write each element for (const auto& el : *j.m_data.m_value.array) @@ -19841,7 +19717,7 @@ class binary_writer write_number(static_cast(0xda)); write_number(static_cast(j.m_data.m_value.binary->subtype())); } - else if (j.m_data.m_value.binary->subtype() <= (std::numeric_limits::max)()) + else { write_number(static_cast(0xdb)); write_number(static_cast(j.m_data.m_value.binary->subtype())); @@ -19850,32 +19726,7 @@ class binary_writer // step 1: write control byte and the binary array size const auto N = j.m_data.m_value.binary->size(); - if (N <= 0x17) - { - write_number(static_cast(0x40 + N)); - } - else if (N <= (std::numeric_limits::max)()) - { - oa.write_character(to_char_type(0x58)); - write_number(static_cast(N)); - } - else if (N <= (std::numeric_limits::max)()) - { - oa.write_character(to_char_type(0x59)); - write_number(static_cast(N)); - } - else if (N <= (std::numeric_limits::max)()) - { - oa.write_character(to_char_type(0x5A)); - write_number(static_cast(N)); - } - // LCOV_EXCL_START - else if (N <= (std::numeric_limits::max)()) - { - oa.write_character(to_char_type(0x5B)); - write_number(static_cast(N)); - } - // LCOV_EXCL_STOP + write_cbor_head(0x40, N); // step 2: write each element oa.write_characters( @@ -19888,33 +19739,7 @@ class binary_writer case value_t::object: { // step 1: write control byte and the object size - const auto N = j.m_data.m_value.object->size(); - if (N <= 0x17) - { - write_number(static_cast(0xA0 + N)); - } - else if (N <= (std::numeric_limits::max)()) - { - oa.write_character(to_char_type(0xB8)); - write_number(static_cast(N)); - } - else if (N <= (std::numeric_limits::max)()) - { - oa.write_character(to_char_type(0xB9)); - write_number(static_cast(N)); - } - else if (N <= (std::numeric_limits::max)()) - { - oa.write_character(to_char_type(0xBA)); - write_number(static_cast(N)); - } - // LCOV_EXCL_START - else if (N <= (std::numeric_limits::max)()) - { - oa.write_character(to_char_type(0xBB)); - write_number(static_cast(N)); - } - // LCOV_EXCL_STOP + write_cbor_head(0xA0, j.m_data.m_value.object->size()); // step 2: write each element for (const auto& el : *j.m_data.m_value.object) @@ -19982,7 +19807,7 @@ class binary_writer oa.write_character(to_char_type(0xCE)); write_number(static_cast(j.m_data.m_value.number_integer)); } - else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) + else { // uint 64 oa.write_character(to_char_type(0xCF)); @@ -20017,8 +19842,7 @@ class binary_writer oa.write_character(to_char_type(0xD2)); write_number(static_cast(j.m_data.m_value.number_integer)); } - else if (j.m_data.m_value.number_integer >= (std::numeric_limits::min)() && - j.m_data.m_value.number_integer <= (std::numeric_limits::max)()) + else { // int 64 oa.write_character(to_char_type(0xD3)); @@ -20053,7 +19877,7 @@ class binary_writer oa.write_character(to_char_type(0xCE)); write_number(static_cast(j.m_data.m_value.number_integer)); } - else if (j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) + else { // uint 64 oa.write_character(to_char_type(0xCF)); @@ -21011,6 +20835,46 @@ class binary_writer // CBOR // ////////// + /*! + @brief write the head of a CBOR data item + + The head is the major type in the upper three bits of the first byte and + an argument - an unsigned integer, the length of a string, the number of + elements of a container - in the shortest of its encodings: in the lower + five bits of the first byte itself if it is at most 23, otherwise in the + 1, 2, 4, or 8 bytes that follow (RFC 8949, section 3). + + @param[in] major_type the major type, shifted into the upper three bits + @param[in] argument the argument of the data item + */ + void write_cbor_head(const std::uint8_t major_type, const std::uint64_t argument) + { + if (argument <= 0x17) + { + write_number(static_cast(major_type + argument)); + } + else if (argument <= (std::numeric_limits::max)()) + { + oa.write_character(to_char_type(static_cast(major_type + 0x18))); + write_number(static_cast(argument)); + } + else if (argument <= (std::numeric_limits::max)()) + { + oa.write_character(to_char_type(static_cast(major_type + 0x19))); + write_number(static_cast(argument)); + } + else if (argument <= (std::numeric_limits::max)()) + { + oa.write_character(to_char_type(static_cast(major_type + 0x1A))); + write_number(static_cast(argument)); + } + else + { + oa.write_character(to_char_type(static_cast(major_type + 0x1B))); + write_number(argument); + } + } + static constexpr CharType get_cbor_float_prefix(float /*unused*/) { return to_char_type(0xFA); // Single-Precision Float @@ -21116,7 +20980,7 @@ class binary_writer } write_number(static_cast(n), use_bjdata); } - else if (use_bjdata && n <= (std::numeric_limits::max)()) + else if (use_bjdata) { if (add_prefix) { @@ -21196,30 +21060,59 @@ class binary_writer } write_number(static_cast(n), use_bjdata); } - else if ((std::numeric_limits::min)() <= n && n <= (std::numeric_limits::max)()) - { - if (add_prefix) - { - oa.write_character(to_char_type('L')); // int64 - } - write_number(static_cast(n), use_bjdata); - } - // LCOV_EXCL_START else { - if (add_prefix) - { - oa.write_character(to_char_type('H')); // high-precision number - } - - const auto number = BasicJsonType(n).dump(); - write_number_with_ubjson_prefix(number.size(), true, use_bjdata); - for (std::size_t i = 0; i < number.size(); ++i) - { - oa.write_character(to_char_type(static_cast(number[i]))); - } + // every value of an integer type of at most 64 bits fits into an + // int64; only a wider type needs a range check + write_ubjson_int64_or_high_precision(n, add_prefix, use_bjdata, + std::integral_constant < bool, std::numeric_limits::digits <= std::numeric_limits::digits > {}); } - // LCOV_EXCL_STOP + } + + template + void write_ubjson_int64_or_high_precision(const NumberType n, const bool add_prefix, const bool use_bjdata, std::true_type /*fits_int64*/) + { + if (add_prefix) + { + oa.write_character(to_char_type('L')); // int64 + } + write_number(static_cast(n), use_bjdata); + } + + template + void write_ubjson_int64_or_high_precision(const NumberType n, const bool add_prefix, const bool use_bjdata, std::false_type /*fits_int64*/) + { + if ((std::numeric_limits::min)() <= n && n <= (std::numeric_limits::max)()) + { + write_ubjson_int64_or_high_precision(n, add_prefix, use_bjdata, std::true_type {}); + return; + } + + if (add_prefix) + { + oa.write_character(to_char_type('H')); // high-precision number + } + + const auto number = BasicJsonType(n).dump(); + write_number_with_ubjson_prefix(number.size(), true, use_bjdata); + for (std::size_t i = 0; i < number.size(); ++i) + { + oa.write_character(to_char_type(static_cast(number[i]))); + } + } + + template + static constexpr CharType ubjson_int64_or_high_precision_prefix(const NumberType /*n*/, std::true_type /*fits_int64*/) noexcept + { + return 'L'; + } + + template + static CharType ubjson_int64_or_high_precision_prefix(const NumberType n, std::false_type /*fits_int64*/) noexcept + { + // anything outside of the range of an int64 is treated as a + // high-precision number + return ((std::numeric_limits::min)() <= n && n <= (std::numeric_limits::max)()) ? 'L' : 'H'; } /*! @@ -21261,12 +21154,10 @@ class binary_writer { return 'm'; } - if ((std::numeric_limits::min)() <= j.m_data.m_value.number_integer && j.m_data.m_value.number_integer <= (std::numeric_limits::max)()) - { - return 'L'; - } - // anything else is treated as a high-precision number - return 'H'; // LCOV_EXCL_LINE + // every value of an integer type of at most 64 bits fits into + // an int64; only a wider type needs a range check + return ubjson_int64_or_high_precision_prefix(j.m_data.m_value.number_integer, + std::integral_constant < bool, std::numeric_limits::digits <= std::numeric_limits::digits > {}); } case value_t::number_unsigned: @@ -21299,12 +21190,12 @@ class binary_writer { return 'L'; } - if (use_bjdata && j.m_data.m_value.number_unsigned <= (std::numeric_limits::max)()) + if (use_bjdata) { return 'M'; } // anything else is treated as a high-precision number - return 'H'; // LCOV_EXCL_LINE + return 'H'; } case value_t::number_float: @@ -27135,17 +27026,6 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec // value access // ////////////////// - /// get a boolean (explicit) - boolean_t get_impl(boolean_t* /*unused*/) const - { - if (JSON_HEDLEY_LIKELY(is_boolean())) - { - return m_data.m_value.boolean; - } - - JSON_THROW(type_error::create(302, detail::concat("type must be boolean, but is ", type_name()), this)); - } - /// get a pointer to the value (object) object_t* get_impl_ptr(object_t* /*unused*/) noexcept { diff --git a/tests/src/unit-ubjson.cpp b/tests/src/unit-ubjson.cpp index c8458c44d..9b3b4be17 100644 --- a/tests/src/unit-ubjson.cpp +++ b/tests/src/unit-ubjson.cpp @@ -2981,3 +2981,18 @@ TEST_CASE("UBJSON roundtrips" * doctest::skip()) } } } + +TEST_CASE("UBJSON optimized array of unsigned integers beyond int64") +{ + // UBJSON has no unsigned 64-bit type, so such values are written as + // high-precision numbers - also as the type of an optimized container + const json j = {18446744073709551615ULL, 9223372036854775808ULL}; + const std::vector expected = + { + '[', '$', 'H', '#', 'i', 2, + 'i', 20, '1', '8', '4', '4', '6', '7', '4', '4', '0', '7', '3', '7', '0', '9', '5', '5', '1', '6', '1', '5', + 'i', 19, '9', '2', '2', '3', '3', '7', '2', '0', '3', '6', '8', '5', '4', '7', '7', '5', '8', '0', '8' + }; + CHECK(json::to_ubjson(j, true, true) == expected); + CHECK(json::from_ubjson(expected) == j); +} From 85f8b21e1ca80137459b9a8cb7180e8acac6ceda Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Sun, 27 Sep 2026 14:19:22 +0200 Subject: [PATCH 57/64] Add tests for uncovered code paths (#5581) * Add tests for uncovered code paths Cover code the test suite did not reach, found from the Coveralls report of develop and a local coverage run of HEAD: - dump() of every kind of value below the bound of the recursive descent (pretty-printed objects, binary values, discarded values, scalars), and flushes of the escape and write buffers mid-string and mid-binary - the iterative comparison: objects with different keys, containers that are a prefix of each other, and elements that cannot be ordered, each both at the top level and below the nesting bound - SAX handlers that stop at any event, including the end of a nested container, in the BSON, CBOR, MessagePack, UBJSON and BJData readers - from_bson/cbor/msgpack/ubjson/bjdata returning a discarded value through the iterator and pointer overloads - JSON Patch, diff, merge_patch and update(..., true) on ordered_json - smaller gaps: get_allocator(), to_ubjson/to_bjdata into a string, value() with an unresolvable JSON pointer, integer/float comparison below the integer range and with negative fractions, conversion to a custom binary type, std::formatter::parse on a spec without '}', unescape() of a lone '~', and the callback parser's start_array() Signed-off-by: Niels Lohmann * Cover more paths that were thought unreachable - parse_float_fast() declining malformed or inexact input, called directly since the lexer only passes well-formed numbers to it - a UTF-16 high surrogate followed by a unit above the low surrogates - self-assignment of a const_iterator - a truncated CBOR string read through non-contiguous iterators - serializing a long double under the de_DE locale, which undoes the locale's decimal point and thousands separator - values read from a binary format carrying no diagnostic positions, with and without a parser callback Signed-off-by: Niels Lohmann * Fix the CI failures of the new coverage tests - declare the self-assignment reference const (misc-const-correctness) - expect the (/path) prefix that JSON_DIAGNOSTICS adds to the messages of the failing ordered_json patch operations Signed-off-by: Niels Lohmann * Expect the byte range JSON_DIAGNOSTIC_POSITIONS adds to the patch errors Signed-off-by: Niels Lohmann * Build the expected dump of the nested-object test with += clang-tidy (performance-inefficient-string-concatenation) reported the chain of operator+ calls that assembled the expected indented output. Signed-off-by: Niels Lohmann * Compare the BJData and UBJSON test outputs byte by byte Building a std::string from the byte vector converts each byte implicitly, which -fsanitize=integer reports for bytes of 0x80 and above (ci_test_clang_sanitizer). Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- tests/src/unit-allocator.cpp | 6 + tests/src/unit-bjdata.cpp | 113 +++++++++++++++++ tests/src/unit-bson.cpp | 38 ++++++ tests/src/unit-cbor.cpp | 47 +++++++ tests/src/unit-class_const_iterator.cpp | 7 ++ tests/src/unit-class_lexer.cpp | 43 +++++++ tests/src/unit-comparison.cpp | 90 +++++++++++++ tests/src/unit-custom-binary-type.cpp | 10 ++ tests/src/unit-diagnostic-positions.cpp | 35 ++++++ tests/src/unit-element_access2.cpp | 10 ++ tests/src/unit-json_patch.cpp | 95 ++++++++++++++ tests/src/unit-json_pointer.cpp | 13 ++ tests/src/unit-locale-cpp.cpp | 11 ++ tests/src/unit-merge_patch.cpp | 29 +++++ tests/src/unit-msgpack.cpp | 38 ++++++ tests/src/unit-serialization.cpp | 160 ++++++++++++++++++++++++ tests/src/unit-std-format.cpp | 23 ++++ tests/src/unit-ubjson.cpp | 63 ++++++++++ tests/src/unit-wstring.cpp | 4 + 19 files changed, 835 insertions(+) diff --git a/tests/src/unit-allocator.cpp b/tests/src/unit-allocator.cpp index cdc532e3a..9dde143c2 100644 --- a/tests/src/unit-allocator.cpp +++ b/tests/src/unit-allocator.cpp @@ -37,6 +37,12 @@ struct bad_allocator : std::allocator }; } // namespace +TEST_CASE("get_allocator") +{ + const auto alloc = nlohmann::json::get_allocator(); + CHECK(alloc == std::allocator()); +} + TEST_CASE("bad_alloc") { SECTION("bad_alloc") diff --git a/tests/src/unit-bjdata.cpp b/tests/src/unit-bjdata.cpp index ebbbbfaf6..03c69431b 100644 --- a/tests/src/unit-bjdata.cpp +++ b/tests/src/unit-bjdata.cpp @@ -3763,6 +3763,49 @@ TEST_CASE("BJData") } } +TEST_CASE("BJData input that cannot be read is discarded by every overload") +{ + std::vector input = json::to_bjdata(json({{"a", {1, 2}}})); + input.pop_back(); + + json _; + CHECK_THROWS_AS(_ = json::from_bjdata(input.begin(), input.end()), json::parse_error&); + CHECK(json::from_bjdata(input, true, false).is_discarded()); + CHECK(json::from_bjdata(input.begin(), input.end(), true, false).is_discarded()); +} + +TEST_CASE("BJData SAX parsing stops at every event") +{ + // Containers are opened and closed by the loop that reads them; a SAX + // handler that rejects any event - including the end of a nested + // container - must stop the parse right there. + const auto count_events = [](const std::vector& input) + { + int events = 0; + while (true) + { + SaxCountdown scp(events); + if (json::sax_parse(input, &scp, json::input_format_t::bjdata)) + { + return events; + } + ++events; + REQUIRE(events < 1000); + } + }; + + // 20 events: every container kind closes inside another one + const json j = json::parse(R"({"a": [1, {"b": []}], "c": {"d": [[2]]}})"); + CHECK(count_events(json::to_bjdata(j)) == 20); + CHECK(count_events(json::to_bjdata(j, true)) == 20); + CHECK(count_events(json::to_bjdata(j, true, true)) == 20); + + // an ND-array is announced as an annotated object: start_object, then + // _ArrayType_, _ArraySize_ and _ArrayData_ with its elements + const json ndarray = json::parse(R"({"_ArrayType_": "uint8", "_ArraySize_": [2, 2], "_ArrayData_": [1, 2, 3, 4]})"); + CHECK(count_events(json::to_bjdata(ndarray, true, true)) == 16); +} + TEST_CASE("issue #5405 - array reserve for definite-length BJData arrays") { #if !defined(JSON_NOEXCEPTION) @@ -4247,6 +4290,65 @@ TEST_CASE("all BJData first bytes") } #endif +TEST_CASE("BJData and UBJSON can be written to a string") +{ + const std::vector values = + { + {{"a", {1, 2.5, "x", nullptr}}, {"b", json::binary({1, 2})}}, + // an annotated ND-array, and objects that only look like one + json::parse(R"({"_ArrayType_": "uint8", "_ArraySize_": [2, 2], "_ArrayData_": [1, 2, 3, 4]})"), + json::parse(R"({"_ArrayType_": 1, "_ArraySize_": [2, 2], "_ArrayData_": [1, 2, 3, 4]})"), + json::parse(R"({"_ArrayType_": "uint8", "_ArraySize_": 4, "_ArrayData_": [1, 2, 3, 4]})"), + json::parse(R"({"_ArrayType_": "uint8", "_ArraySize_": [2, -2], "_ArrayData_": [1, 2, 3, 4]})"), + json::parse(R"({"_ArrayType_": "uint8", "_ArraySize_": [2, 2], "_ArrayData_": [1, 2, 3]})"), + json::parse(R"({"_ArrayType_": "uint8", "_ArraySize_": [2, 2], "_ArrayData_": 1})"), + }; + + // compared byte by byte: building a std::string from the bytes would + // convert them implicitly, which -fsanitize=integer reports for bytes of + // 0x80 and above + const auto same_bytes = [](const std::vector& bytes, const std::string & text) + { + return bytes.size() == text.size() && std::equal(bytes.begin(), bytes.end(), text.begin(), [](std::uint8_t byte, char c) + { + return byte == static_cast(c); + }); + }; + + for (const auto& j : values) + { + CAPTURE(j.dump()); + for (const bool use_size : + { + false, true + }) + { + for (const bool use_type : + { + false, true + }) + { + if (use_type && !use_size) + { + continue; + } + CAPTURE(use_size); + CAPTURE(use_type); + + const auto bjdata = json::to_bjdata(j, use_size, use_type); + std::string bjdata_string; + json::to_bjdata(j, bjdata_string, use_size, use_type); + CHECK(same_bytes(bjdata, bjdata_string)); + + const auto ubjson = json::to_ubjson(j, use_size, use_type); + std::string ubjson_string; + json::to_ubjson(j, ubjson_string, use_size, use_type); + CHECK(same_bytes(ubjson, ubjson_string)); + } + } + } +} + TEST_CASE("BJData use_type requires use_size") { SECTION("non-empty object throws other_error.502") @@ -4265,6 +4367,17 @@ TEST_CASE("BJData use_type requires use_size") json::other_error&); } + SECTION("non-empty binary value throws other_error.502") + { + const json j = json::binary({1, 2, 3}); + CHECK_THROWS_WITH_AS(json::to_bjdata(j, false, true), + "[json.exception.other_error.502] use_type requires use_size = true", + json::other_error&); + CHECK_THROWS_WITH_AS(json::to_ubjson(j, false, true), + "[json.exception.other_error.502] use_type requires use_size = true", + json::other_error&); + } + SECTION("scalars do not throw with use_type=true, use_count=false") { CHECK_NOTHROW(json::to_bjdata(42, false, true)); diff --git a/tests/src/unit-bson.cpp b/tests/src/unit-bson.cpp index 03cde5376..81331b43a 100644 --- a/tests/src/unit-bson.cpp +++ b/tests/src/unit-bson.cpp @@ -1244,6 +1244,44 @@ TEST_CASE("BSON nesting does not consume the call stack") } } +TEST_CASE("BSON input that cannot be read is discarded by every overload") +{ + std::vector input = json::to_bson(json({{"a", {1, 2}}})); + input.pop_back(); + + json _; + CHECK_THROWS_AS(_ = json::from_bson(input.begin(), input.end()), json::parse_error&); + CHECK(json::from_bson(input, true, false).is_discarded()); + CHECK(json::from_bson(input.begin(), input.end(), true, false).is_discarded()); + CHECK(json::from_bson(input.data(), input.size(), true, false).is_discarded()); + CHECK(json::from_bson({input.data(), input.size()}, true, false).is_discarded()); +} + +TEST_CASE("BSON SAX parsing stops at every event") +{ + // Containers are opened and closed by the loop that reads them; a SAX + // handler that rejects any event - including the end of a nested + // container - must stop the parse right there. + const auto count_events = [](const std::vector& input) + { + int events = 0; + while (true) + { + SaxCountdown scp(events); + if (json::sax_parse(input, &scp, json::input_format_t::bson)) + { + return events; + } + ++events; + REQUIRE(events < 1000); + } + }; + + // 20 events: every container kind closes inside another one + const json j = json::parse(R"({"a": [1, {"b": []}], "c": {"d": [[2]]}})"); + CHECK(count_events(json::to_bson(j)) == 20); +} + TEST_CASE("BSON numerical data") { SECTION("number") diff --git a/tests/src/unit-cbor.cpp b/tests/src/unit-cbor.cpp index 9c799b0eb..353281d5e 100644 --- a/tests/src/unit-cbor.cpp +++ b/tests/src/unit-cbor.cpp @@ -15,6 +15,7 @@ using nlohmann::json; #include #include #include +#include #include #include "make_test_data_available.hpp" #include "test_utils.hpp" @@ -2175,6 +2176,52 @@ TEST_CASE("CBOR nesting does not consume the call stack") } } +TEST_CASE("CBOR input that cannot be read is discarded by every overload") +{ + std::vector input = json::to_cbor(json({{"a", {1, 2}}})); + input.pop_back(); + + json _; + CHECK_THROWS_AS(_ = json::from_cbor(input.begin(), input.end()), json::parse_error&); + CHECK(json::from_cbor(input, true, false).is_discarded()); + CHECK(json::from_cbor(input.begin(), input.end(), true, false).is_discarded()); + CHECK(json::from_cbor(input.data(), input.size(), true, false).is_discarded()); + CHECK(json::from_cbor({input.data(), input.size()}, true, false).is_discarded()); + + // a string that ends early, read through iterators that are not + // contiguous and have to be copied from one element at a time + const std::list truncated_string = {0x63, 'a', 'b'}; + CHECK(json::from_cbor(truncated_string.begin(), truncated_string.end(), true, false).is_discarded()); + const std::list complete_string = {0x63, 'a', 'b', 'c'}; + CHECK(json::from_cbor(complete_string.begin(), complete_string.end()) == "abc"); +} + +TEST_CASE("CBOR SAX parsing stops at every event") +{ + // Containers are opened and closed by the loop that reads them; a SAX + // handler that rejects any event - including the end of a nested + // container - must stop the parse right there. + const auto count_events = [](const std::vector& input) + { + int events = 0; + while (true) + { + SaxCountdown scp(events); + if (json::sax_parse(input, &scp, json::input_format_t::cbor)) + { + return events; + } + ++events; + REQUIRE(events < 1000); + } + }; + + // 20 events: every container kind closes inside another one + const json j = json::parse(R"({"a": [1, {"b": []}], "c": {"d": [[2]]}})"); + CHECK(count_events(json::to_cbor(j)) == 20); + CHECK(count_events(std::vector({0xBF, 0x61, 'a', 0x9F, 0x01, 0xFF, 0xFF})) == 6); +} + TEST_CASE("CBOR indefinite-length strings do not recurse per chunk") { // Reading an indefinite-length string or byte array used to call itself diff --git a/tests/src/unit-class_const_iterator.cpp b/tests/src/unit-class_const_iterator.cpp index 9e5de42c7..e21e18fc9 100644 --- a/tests/src/unit-class_const_iterator.cpp +++ b/tests/src/unit-class_const_iterator.cpp @@ -43,6 +43,13 @@ TEST_CASE("const_iterator class") json::const_iterator const it(&j); json::const_iterator it2(&j); it2 = it; + + // assigning an iterator to itself leaves it unchanged + json const a = {1, 2, 3}; + json::const_iterator it3 = a.cbegin() + 1; + const json::const_iterator& same = it3; + it3 = same; + CHECK(*it3 == 2); } SECTION("copy constructor from non-const iterator") diff --git a/tests/src/unit-class_lexer.cpp b/tests/src/unit-class_lexer.cpp index e89497738..cb8ceed6b 100644 --- a/tests/src/unit-class_lexer.cpp +++ b/tests/src/unit-class_lexer.cpp @@ -12,6 +12,7 @@ #include using nlohmann::json; +#include // FLT_EVAL_METHOD #include // strtod #include // stringstream #include // string @@ -657,3 +658,45 @@ TEST_CASE("lexer string fast path") } } } + +TEST_CASE("parse_float_fast declines what it cannot convert exactly") +{ + // The lexer only hands well-formed numbers to parse_float_fast, so the + // malformed ones below can only be passed to it directly. Declining is + // always safe: the caller then falls back to a slower, exact conversion. + const auto fast = [](const std::string & s, double & out) + { + return nlohmann::detail::parse_float_fast(s.data(), s.data() + s.size(), '.', out); + }; + double out = 0; + +#if defined(FLT_EVAL_METHOD) && FLT_EVAL_METHOD != 0 + // without true double precision, the fast path declines everything + CHECK_FALSE(fast("1.5", out)); +#else + CHECK(fast("1.5", out)); + CHECK(out == 1.5); + CHECK(fast("+2.5e1", out)); + CHECK(out == 25.0); + CHECK(fast("-25E-1", out)); + CHECK(out == -2.5); + CHECK(fast("1e", out)); + CHECK(out == 1.0); +#endif + + // not a number + CHECK_FALSE(fast("", out)); + CHECK_FALSE(fast("-", out)); + CHECK_FALSE(fast(".", out)); + CHECK_FALSE(fast("1.2.3", out)); + CHECK_FALSE(fast("1x", out)); + CHECK_FALSE(fast("1e+", out)); + CHECK_FALSE(fast("1e1x", out)); + + // numbers that are not represented exactly on the fast path + CHECK_FALSE(fast("12345678901234567890", out)); + CHECK_FALSE(fast("1e10000", out)); + CHECK_FALSE(fast("9007199254740993", out)); + CHECK_FALSE(fast("1e23", out)); + CHECK_FALSE(fast("1e-23", out)); +} diff --git a/tests/src/unit-comparison.cpp b/tests/src/unit-comparison.cpp index 68d20aca9..9a6606256 100644 --- a/tests/src/unit-comparison.cpp +++ b/tests/src/unit-comparison.cpp @@ -359,6 +359,15 @@ TEST_CASE("lexicographical comparison operators") CHECK(json(1) < json(1.5)); CHECK(json(1.5) < json(2)); CHECK(json(2) > json(1.5)); + CHECK(json(-1) > json(-1.5)); + CHECK(json(-1.5) < json(-1)); + CHECK(json(-2) < json(-1.5)); + + // a float below the range of the integer type + CHECK(json(0) > json(-1e30)); + CHECK(json(-1e30) < json(0)); + CHECK(json(0u) > json(-0.5)); + CHECK(json(-0.5) < json(0u)); // a NaN operand stays unordered against either integer kind CHECK_FALSE(json(1) == json(nan)); @@ -735,3 +744,84 @@ TEST_CASE("regression #3868 - heterogeneous comparisons compile under C++20 (P24 } } #endif + +TEST_CASE("containers are compared element by element") +{ + // Containers nested deeper than a bound are compared without the call + // stack, by code of their own; every relation is checked both at the top + // level and below that bound. + const auto deep = [](const json & j, const std::size_t depth) + { + json result = j; + for (std::size_t i = 0; i < depth; ++i) + { + result = json::array({std::move(result)}); + } + return result; + }; + + for (const std::size_t depth : std::vector {0, 200}) + { + CAPTURE(depth); + + // objects with different keys + { + const json a = deep({{"a", 1}}, depth); + const json b = deep({{"b", 1}}, depth); + CHECK_FALSE(a == b); + CHECK(a != b); + CHECK(a < b); + CHECK(b > a); + CHECK_FALSE(b < a); +#if JSON_HAS_THREE_WAY_COMPARISON + // JSON_HAS_CPP_20 (do not remove; see note at top of file) + CHECK((a <=> b) == std::partial_ordering::less); // *NOPAD* + CHECK((b <=> a) == std::partial_ordering::greater); // *NOPAD* + CHECK((a <=> a) == std::partial_ordering::equivalent); // *NOPAD* +#endif + } + + // a container that is a prefix of the other one + { + // the one that runs out of elements first is the smaller one + const json shorter = deep({1}, depth); + const json longer = deep({1, 2}, depth); + CHECK(shorter < longer); + CHECK(longer > shorter); + CHECK_FALSE(longer < shorter); + CHECK_FALSE(shorter == longer); + + const json smaller_object = deep({{"a", 1}}, depth); + const json larger_object = deep({{"a", 1}, {"b", 2}}, depth); + CHECK(smaller_object < larger_object); + CHECK(larger_object > smaller_object); + CHECK_FALSE(smaller_object == larger_object); +#if JSON_HAS_THREE_WAY_COMPARISON + // JSON_HAS_CPP_20 (do not remove; see note at top of file) + CHECK((shorter <=> longer) == std::partial_ordering::less); // *NOPAD* + CHECK((longer <=> shorter) == std::partial_ordering::greater); // *NOPAD* +#endif + } + + // elements that cannot be ordered + { + const double nan = std::numeric_limits::quiet_NaN(); + const json lhs = deep({nan, 1}, depth); + const json rhs = deep({nan, 2}, depth); + + CHECK_FALSE(lhs == lhs); + CHECK_FALSE(rhs < lhs); +#if JSON_HAS_THREE_WAY_COMPARISON + // JSON_HAS_CPP_20 (do not remove; see note at top of file) + // operator<=> stops there, as std::lexicographical_compare_three_way + // does, and operator< is derived from it + CHECK((lhs <=> rhs) == std::partial_ordering::unordered); // *NOPAD* + CHECK_FALSE(lhs < rhs); +#else + // operator< skips a pair of elements that cannot be ordered, as + // std::lexicographical_compare does, and the next pair decides + CHECK(lhs < rhs); +#endif + } + } +} diff --git a/tests/src/unit-custom-binary-type.cpp b/tests/src/unit-custom-binary-type.cpp index d357ec9a3..efedba3cd 100644 --- a/tests/src/unit-custom-binary-type.cpp +++ b/tests/src/unit-custom-binary-type.cpp @@ -49,6 +49,16 @@ TEST_CASE("binary type whose value type is not std::uint8_t") CHECK(char_binary_json::binary({}).dump() == R"({"bytes":[],"subtype":null})"); } + SECTION("a value is converted to the binary type if it is binary or an array") + { + const std::vector chars{'\0', '\x01', '\x7F'}; + CHECK(char_binary_json::binary(chars).get>() == chars); + CHECK(char_binary_json({0, 1, 127}).get>() == chars); + CHECK_THROWS_WITH_AS(char_binary_json(1).get>(), + "[json.exception.type_error.302] type must be binary or array, but is number", + char_binary_json::type_error&); + } + SECTION("the default binary type is unchanged") { CHECK(nlohmann::json::binary({0, 1, 255}, 42).dump() == R"({"bytes":[0,1,255],"subtype":42})"); diff --git a/tests/src/unit-diagnostic-positions.cpp b/tests/src/unit-diagnostic-positions.cpp index d607e935c..5326094c7 100644 --- a/tests/src/unit-diagnostic-positions.cpp +++ b/tests/src/unit-diagnostic-positions.cpp @@ -156,3 +156,38 @@ TEST_CASE("Better diagnostics with positions") #endif } } + +TEST_CASE("values read from a binary format have no positions") +{ + // only the JSON lexer knows where a value started and ended + const json source = {{"a", {1, "x", json::binary({1})}}, {"b", {{"c", true}}}, {"d", nullptr}, {"e", 1.5}}; + const std::vector cbor = json::to_cbor(source); + + const auto check_no_positions = [](const json & j) + { + CHECK(j.start_pos() == std::string::npos); + CHECK(j.end_pos() == std::string::npos); + CHECK(j.at("a").start_pos() == std::string::npos); + CHECK(j.at("a").at(1).end_pos() == std::string::npos); + CHECK(j.at("b").at("c").start_pos() == std::string::npos); + }; + + SECTION("DOM parser") + { + const json j = json::from_cbor(cbor); + CHECK(j == source); + check_no_positions(j); + } + + SECTION("DOM parser with a callback") + { + json j; + nlohmann::detail::json_sax_dom_callback_parser sdp(j, [](int /*unused*/, json::parse_event_t /*unused*/, const json& /*unused*/) noexcept + { + return true; + }); + CHECK(json::sax_parse(cbor, &sdp, json::input_format_t::cbor)); + CHECK(j == source); + check_no_positions(j); + } +} diff --git a/tests/src/unit-element_access2.cpp b/tests/src/unit-element_access2.cpp index c7c1b824d..40e8216d5 100644 --- a/tests/src/unit-element_access2.cpp +++ b/tests/src/unit-element_access2.cpp @@ -1517,6 +1517,16 @@ TEST_CASE_TEMPLATE("element access 2 (throwing tests)", Json, nlohmann::json, nl CHECK(j.value("/not/existing"_json_pointer, Json({{"foo", "bar"}})) == Json({{"foo", "bar"}})); CHECK(j.value("/not/existing"_json_pointer, Json({10, 100})) == Json({10, 100})); + // an array index that is out of range, too large to be + // represented, or "-", and a token below a scalar + CHECK(j.value("/array/3"_json_pointer, 2) == 2); + CHECK(j.value("/array/-"_json_pointer, 2) == 2); + CHECK(j.value("/array/99999999999999999999999999"_json_pointer, 2) == 2); + CHECK(j.value("/integer/0"_json_pointer, 2) == 2); + CHECK(j.value("/string/x"_json_pointer, 2) == 2); + CHECK(j.value("/null/x"_json_pointer, 2) == 2); + CHECK(j.value("/array/0"_json_pointer, 2) == 1); + CHECK(j_const.value("/not/existing"_json_pointer, 2) == 2); CHECK(j_const.value("/not/existing"_json_pointer, 2u) == 2u); CHECK(j_const.value("/not/existing"_json_pointer, false) == false); diff --git a/tests/src/unit-json_patch.cpp b/tests/src/unit-json_patch.cpp index 7731c7d92..216d00c41 100644 --- a/tests/src/unit-json_patch.cpp +++ b/tests/src/unit-json_patch.cpp @@ -1751,3 +1751,98 @@ TEST_CASE("JSON patch - diff emits array removals in descending index order") CHECK(source.patch(patch) == target); } } + +TEST_CASE("JSON patch - every operation on ordered_json") +{ + using nlohmann::ordered_json; + + const ordered_json doc = {{"foo", "bar"}, {"arr", {1, 2, 3}}, {"obj", {{"a", 1}}}}; + + SECTION("successful operations") + { + const ordered_json patch = ordered_json::parse(R"([ + {"op": "add", "path": "/obj/b", "value": 2}, + {"op": "add", "path": "/arr/1", "value": 9}, + {"op": "add", "path": "/arr/-", "value": 4}, + {"op": "remove", "path": "/arr/0"}, + {"op": "remove", "path": "/obj/a"}, + {"op": "replace", "path": "/foo", "value": "baz"}, + {"op": "move", "from": "/foo", "path": "/moved"}, + {"op": "copy", "from": "/obj", "path": "/copied"}, + {"op": "test", "path": "/copied/b", "value": 2} + ])"); + + const ordered_json expected = ordered_json::parse(R"({ + "arr": [9, 2, 3, 4], "obj": {"b": 2}, "moved": "baz", "copied": {"b": 2} + })"); + + CHECK(doc.patch(patch) == expected); + + // adding to the root replaces the document + CHECK(doc.patch(ordered_json::parse(R"([{"op": "add", "path": "", "value": [1]}])")) == ordered_json({1})); + } + + SECTION("failing operations") + { + ordered_json _; +#if JSON_DIAGNOSTICS + CHECK_THROWS_WITH_AS(_ = doc.patch(ordered_json::parse(R"([{"op": "add", "path": "/arr/4", "value": 1}])")), + "[json.exception.out_of_range.401] (/arr) array index 4 is out of range", ordered_json::out_of_range&); +#else + CHECK_THROWS_WITH_AS(_ = doc.patch(ordered_json::parse(R"([{"op": "add", "path": "/arr/4", "value": 1}])")), + "[json.exception.out_of_range.401] array index 4 is out of range", ordered_json::out_of_range&); +#endif + CHECK_THROWS_WITH_AS(_ = doc.patch(ordered_json::parse(R"([{"op": "add", "path": "/nope/x", "value": 1}])")), + "[json.exception.out_of_range.403] key 'nope' not found", ordered_json::out_of_range&); + CHECK_THROWS_WITH_AS(_ = doc.patch(ordered_json::parse(R"([{"op": "remove", "path": "/obj/nope"}])")), + "[json.exception.out_of_range.403] key 'nope' not found", ordered_json::out_of_range&); +#if JSON_DIAGNOSTICS + CHECK_THROWS_WITH_AS(_ = doc.patch(ordered_json::parse(R"([{"op": "remove", "path": "/arr/3"}])")), + "[json.exception.out_of_range.401] (/arr) array index 3 is out of range", ordered_json::out_of_range&); +#else + CHECK_THROWS_WITH_AS(_ = doc.patch(ordered_json::parse(R"([{"op": "remove", "path": "/arr/3"}])")), + "[json.exception.out_of_range.401] array index 3 is out of range", ordered_json::out_of_range&); +#endif +#if JSON_DIAGNOSTICS + CHECK_THROWS_WITH_AS(_ = doc.patch(ordered_json::parse(R"([{"op": "test", "path": "/foo", "value": "qux"}])")), + "[json.exception.other_error.501] (/0) unsuccessful: {\"op\":\"test\",\"path\":\"/foo\",\"value\":\"qux\"}", ordered_json::other_error&); +#elif JSON_DIAGNOSTIC_POSITIONS + CHECK_THROWS_WITH_AS(_ = doc.patch(ordered_json::parse(R"([{"op": "test", "path": "/foo", "value": "qux"}])")), + "[json.exception.other_error.501] (bytes 1-47) unsuccessful: {\"op\":\"test\",\"path\":\"/foo\",\"value\":\"qux\"}", ordered_json::other_error&); +#else + CHECK_THROWS_WITH_AS(_ = doc.patch(ordered_json::parse(R"([{"op": "test", "path": "/foo", "value": "qux"}])")), + "[json.exception.other_error.501] unsuccessful: {\"op\":\"test\",\"path\":\"/foo\",\"value\":\"qux\"}", ordered_json::other_error&); +#endif +#if JSON_DIAGNOSTICS + CHECK_THROWS_WITH_AS(_ = doc.patch(ordered_json::parse(R"([{"op": "add", "path": "/foo"}])")), + "[json.exception.parse_error.105] parse error: (/0) operation 'add' must have member 'value'", ordered_json::parse_error&); +#elif JSON_DIAGNOSTIC_POSITIONS + CHECK_THROWS_WITH_AS(_ = doc.patch(ordered_json::parse(R"([{"op": "add", "path": "/foo"}])")), + "[json.exception.parse_error.105] parse error: (bytes 1-30) operation 'add' must have member 'value'", ordered_json::parse_error&); +#else + CHECK_THROWS_WITH_AS(_ = doc.patch(ordered_json::parse(R"([{"op": "add", "path": "/foo"}])")), + "[json.exception.parse_error.105] parse error: operation 'add' must have member 'value'", ordered_json::parse_error&); +#endif + CHECK_THROWS_WITH_AS(_ = doc.patch(ordered_json::parse(R"([{"op": "move", "from": "/obj", "path": "/obj/a/b"}])")), + "[json.exception.out_of_range.414] cannot move value: 'from' path '/obj' is a proper prefix of 'path' '/obj/a/b'", ordered_json::out_of_range&); + } + + SECTION("diff reproduces the target") + { + const ordered_json source = {{"a", 1}, {"b", 2}, {"c", {{"x", 1}}}, {"l", {1, 2, 3}}}; + const std::vector targets = + { + // a key removed, a key added, a nested change, a shorter array + {{"a", 1}, {"c", {{"x", 2}}}, {"l", {1}}, {"d", 4}}, + // the same keys in another order + {{"c", {{"x", 1}}}, {"a", 1}, {"b", 2}, {"l", {1, 2, 3}}}, + // new keys ahead of the common ones + {{"new", true}, {"a", 1}, {"b", 3}, {"c", {{"x", 1}}}, {"l", {1, 2, 3}}}, + }; + for (const auto& target : targets) + { + CAPTURE(target.dump()); + CHECK(source.patch(ordered_json::diff(source, target)) == target); + } + } +} diff --git a/tests/src/unit-json_pointer.cpp b/tests/src/unit-json_pointer.cpp index b01df2921..e7d6df530 100644 --- a/tests/src/unit-json_pointer.cpp +++ b/tests/src/unit-json_pointer.cpp @@ -872,3 +872,16 @@ TEST_CASE("JSON pointers") } #endif } + +TEST_CASE("unescaping keeps a '~' that does not start an escape sequence") +{ + // the parser of a JSON pointer rejects such reference tokens before it + // unescapes them, so this is only reachable by calling unescape directly + std::string s = "a~2b~"; + nlohmann::detail::unescape(s); + CHECK(s == "a~2b~"); + + s = "~0~1~"; + nlohmann::detail::unescape(s); + CHECK(s == "~/~"); +} diff --git a/tests/src/unit-locale-cpp.cpp b/tests/src/unit-locale-cpp.cpp index 0019d3b92..c2a113d06 100644 --- a/tests/src/unit-locale-cpp.cpp +++ b/tests/src/unit-locale-cpp.cpp @@ -158,6 +158,17 @@ TEST_CASE("locale-dependent test (LC_NUMERIC=de_DE)") json::sax_parse("12.34", &sax); CHECK(sax.float_string_copy == "12.34"); } + + SECTION("serializing a long double") + { + // a floating-point type that is not a float or a double is written + // with snprintf, whose locale-specific decimal point and thousands + // separator are undone afterwards + using long_double_json = nlohmann::basic_json; + CHECK(long_double_json(12345.5L).dump() == "12345.5"); + CHECK(long_double_json(1.0L).dump() == "1.0"); + CHECK(long_double_json(-0.25L).dump() == "-0.25"); + } } else { diff --git a/tests/src/unit-merge_patch.cpp b/tests/src/unit-merge_patch.cpp index c9741e85b..b1e0c431b 100644 --- a/tests/src/unit-merge_patch.cpp +++ b/tests/src/unit-merge_patch.cpp @@ -345,3 +345,32 @@ TEST_CASE("JSON Merge Patch on deeply nested values") CHECK(p->at("x") == 1); } } + +TEST_CASE("JSON Merge Patch and update on ordered_json") +{ + using nlohmann::ordered_json; + + SECTION("merge_patch") + { + ordered_json target = ordered_json::parse(R"({"a": {"b": 1, "c": 2}, "d": 3, "e": [1]})"); + target.merge_patch(ordered_json::parse(R"({"a": {"b": null, "f": 4}, "d": {"x": {"y": null}}, "e": null, "g": {"h": 5}})")); + CHECK(target == ordered_json::parse(R"({"a": {"c": 2, "f": 4}, "d": {"x": {}}, "g": {"h": 5}})")); + + // a patch that is not an object replaces the target + target.merge_patch(ordered_json({1, 2})); + CHECK(target == ordered_json({1, 2})); + // an object patch turns a target that is not an object into one + target.merge_patch(ordered_json::parse(R"({"k": {"l": null}})")); + CHECK(target == ordered_json::parse(R"({"k": {}})")); + } + + SECTION("update with merge_objects") + { + ordered_json target = ordered_json::parse(R"({"a": {"b": 1, "c": {"d": 2}}, "e": 3})"); + target.update(ordered_json::parse(R"({"a": {"c": {"x": 1}, "f": 4}, "e": {"y": 5}, "g": 6})"), true); + CHECK(target == ordered_json::parse(R"({"a": {"b": 1, "c": {"d": 2, "x": 1}, "f": 4}, "e": {"y": 5}, "g": 6})")); + + target.update(ordered_json::parse(R"({"a": 1})"), false); + CHECK(target == ordered_json::parse(R"({"a": 1, "e": {"y": 5}, "g": 6})")); + } +} diff --git a/tests/src/unit-msgpack.cpp b/tests/src/unit-msgpack.cpp index 8168b88ae..0b8c6ca63 100644 --- a/tests/src/unit-msgpack.cpp +++ b/tests/src/unit-msgpack.cpp @@ -1780,6 +1780,44 @@ TEST_CASE("MessagePack nesting does not consume the call stack") } } +TEST_CASE("MessagePack input that cannot be read is discarded by every overload") +{ + std::vector input = json::to_msgpack(json({{"a", {1, 2}}})); + input.pop_back(); + + json _; + CHECK_THROWS_AS(_ = json::from_msgpack(input.begin(), input.end()), json::parse_error&); + CHECK(json::from_msgpack(input, true, false).is_discarded()); + CHECK(json::from_msgpack(input.begin(), input.end(), true, false).is_discarded()); + CHECK(json::from_msgpack(input.data(), input.size(), true, false).is_discarded()); + CHECK(json::from_msgpack({input.data(), input.size()}, true, false).is_discarded()); +} + +TEST_CASE("MessagePack SAX parsing stops at every event") +{ + // Containers are opened and closed by the loop that reads them; a SAX + // handler that rejects any event - including the end of a nested + // container - must stop the parse right there. + const auto count_events = [](const std::vector& input) + { + int events = 0; + while (true) + { + SaxCountdown scp(events); + if (json::sax_parse(input, &scp, json::input_format_t::msgpack)) + { + return events; + } + ++events; + REQUIRE(events < 1000); + } + }; + + // 20 events: every container kind closes inside another one + const json j = json::parse(R"({"a": [1, {"b": []}], "c": {"d": [[2]]}})"); + CHECK(count_events(json::to_msgpack(j)) == 20); +} + TEST_CASE("single MessagePack roundtrip") { SECTION("sample.json") diff --git a/tests/src/unit-serialization.cpp b/tests/src/unit-serialization.cpp index 45617c3b3..bfb510fd0 100644 --- a/tests/src/unit-serialization.cpp +++ b/tests/src/unit-serialization.cpp @@ -639,3 +639,163 @@ TEST_CASE("serialization of deeply nested values") } } } + +namespace +{ +// wraps @a inner into @a depth single-element arrays +json wrap_in_arrays(const json& inner, const std::size_t depth) +{ + json j = inner; + for (std::size_t i = 0; i < depth; ++i) + { + j = json::array({std::move(j)}); + } + return j; +} + +// what wrap_in_arrays(inner, depth).dump(2) is expected to be: the arrays +// around inner.dump(2), with inner's own lines indented by the depth +std::string expected_pretty_in_arrays(const json& inner, const std::size_t depth) +{ + std::string expected; + for (std::size_t i = 0; i < depth; ++i) + { + expected += std::string(2 * i, ' ') + "[\n"; + } + + const std::string indent(2 * depth, ' '); + expected += indent; + for (const char c : inner.dump(2)) + { + expected += c; + if (c == '\n') + { + expected += indent; + } + } + + for (std::size_t i = depth; i > 0; --i) + { + expected += '\n' + std::string(2 * (i - 1), ' ') + ']'; + } + return expected; +} +} // namespace + +TEST_CASE("serialization of every kind of value below the bound of the descent") +{ + // Values nested deeper than the bound are written without the call stack, + // by code of their own; each kind of value must come out the same there as + // it does at the top level, compact and pretty-printed. + std::vector values = + { + json::parse(R"({"a": 1, "b": [1, 2, {"c": "x"}], "d": {}, "e": []})"), + json::parse(R"([1, [2, 3], {"k": null}, "s"])"), + json::object(), + json::array(), + json::binary({1, 2, 3}, 42), + json::binary({1, 2, 3}), + json::binary({}, 7), + json::binary({}), + "a string with \"escapes\"\n", + true, + false, + -42, + 42u, + 1.5, + nullptr, + json(json::value_t::discarded), + }; + // a pretty-printed object whose members are themselves deep + values.push_back({{"x", wrap_in_arrays(1, 5)}, {"y", {{"z", 2}}}}); + + for (const std::size_t depth : std::vector {1, 200}) + { + CAPTURE(depth); + for (const auto& inner : values) + { + CAPTURE(inner.dump()); + const json j = wrap_in_arrays(inner, depth); + CHECK(j.dump() == std::string(depth, '[') + inner.dump() + std::string(depth, ']')); + CHECK(j.dump(2) == expected_pretty_in_arrays(inner, depth)); + } + } + + SECTION("pretty-printed objects across the bound") + { + for (std::size_t d = 120; d <= 140; ++d) + { + CAPTURE(d); + + // built from the inside out: {"k": , "n": } + json j = 7; + std::string expected = "7"; + for (std::size_t i = d; i > 0; --i) + { + j = json({{"k", std::move(j)}, {"n", i}}); + + const std::string indent(2 * i, ' '); + const std::string outer_indent(2 * (i - 1), ' '); + std::string next = "{\n"; + next += indent; + next += "\"k\": "; + next += expected; + next += ",\n"; + next += indent; + next += "\"n\": "; + next += std::to_string(i); + next += '\n'; + next += outer_indent; + next += '}'; + expected = std::move(next); + } + + CHECK(j.dump(2) == expected); + CHECK(json::parse(j.dump(2)) == j); + CHECK(json::parse(j.dump()) == j); + } + } +} + +TEST_CASE("serializer buffers are flushed mid-string and mid-binary") +{ + SECTION("a long run of escaped characters") + { + // each character is escaped on its own, so the escape buffer fills up + const json newlines = std::string(600, '\n'); + std::string expected = "\""; + for (int i = 0; i < 600; ++i) + { + expected += "\\n"; + } + expected += '"'; + CHECK(newlines.dump() == expected); + + // every character is \u-escaped under ensure_ascii + std::string umlauts; + std::string escaped_umlauts = "\""; + for (int i = 0; i < 300; ++i) + { + umlauts += "\xC3\xA4"; + escaped_umlauts += "\\u00e4"; + } + escaped_umlauts += '"'; + CHECK(json(umlauts).dump(-1, ' ', true) == escaped_umlauts); + } + + SECTION("a large binary value") + { + std::vector bytes(3000); + std::string expected_bytes; + std::string expected_pretty_bytes; + for (std::size_t i = 0; i < bytes.size(); ++i) + { + bytes[i] = static_cast(i % 256); + expected_bytes += (i == 0 ? "" : ",") + std::to_string(i % 256); + expected_pretty_bytes += (i == 0 ? "" : ", ") + std::to_string(i % 256); + } + const json j = json::binary(bytes); + CHECK(j.dump() == "{\"bytes\":[" + expected_bytes + "],\"subtype\":null}"); + CHECK(j.dump(2) == "{\n \"bytes\": [" + expected_pretty_bytes + "],\n \"subtype\": null\n}"); + } +} diff --git a/tests/src/unit-std-format.cpp b/tests/src/unit-std-format.cpp index 58cbbf5cc..f2be8d5ec 100644 --- a/tests/src/unit-std-format.cpp +++ b/tests/src/unit-std-format.cpp @@ -102,6 +102,29 @@ TEST_CASE("std::formatter") CHECK_THROWS_AS(std::vformat("{:{}}", std::make_format_args(j, dynamic_width)), std::format_error); // dynamic width } + SECTION("a format spec may run to the end of the parse context") + { + // std::format always hands parse() a range that still holds the closing + // '}', but a parse context may also end right after the spec + const auto parse = [](const char* spec) + { + std::format_parse_context ctx(spec); + std::formatter f; + CHECK(f.parse(ctx) == ctx.end()); + return f; + }; + + CHECK(parse("").indent == -1); + CHECK(parse(">").indent == -1); + CHECK(parse("#").indent == 4); + CHECK(parse("3").indent == 3); + CHECK(parse("#12").indent == 12); + + const auto f = parse(".>"); + CHECK(f.indent == -1); + CHECK(f.indent_char == '.'); + } + SECTION("std::format_to writes through an arbitrary output iterator") { const json j = {{"foo", 1}, {"bar", {1, 2, 3}}}; diff --git a/tests/src/unit-ubjson.cpp b/tests/src/unit-ubjson.cpp index 9b3b4be17..c8a300e72 100644 --- a/tests/src/unit-ubjson.cpp +++ b/tests/src/unit-ubjson.cpp @@ -1640,6 +1640,29 @@ TEST_CASE("UBJSON") }); CHECK_THROWS_AS(_ = json::sax_parse(v_ubjson, &scp, json::input_format_t::ubjson), json::out_of_range&); } + + SECTION("array with a known size, read with a callback") + { + // a sized array announces its length to start_array() + std::vector const v_ubjson = {'[', '#', 'i', 2, 'i', 1, 'i', 2}; + json j; + nlohmann::detail::json_sax_dom_callback_parser scp(j, [](int /*unused*/, json::parse_event_t /*unused*/, const json& /*unused*/) noexcept + { + return true; + }); + CHECK(json::sax_parse(v_ubjson, &scp, json::input_format_t::ubjson)); + CHECK(j == json({1, 2})); + + // the readers reject a size this large before they announce + // it, so it can only reach start_array() directly (the largest + // value stands for an unknown size and is never checked) + json k; + nlohmann::detail::json_sax_dom_callback_parser scp2(k, [](int /*unused*/, json::parse_event_t /*unused*/, const json& /*unused*/) noexcept + { + return true; + }); + CHECK_THROWS_AS(scp2.start_array((std::numeric_limits::max)() - 1), json::out_of_range&); + } } } @@ -2255,6 +2278,46 @@ TEST_CASE("UBJSON nesting does not consume the call stack") } } +TEST_CASE("UBJSON input that cannot be read is discarded by every overload") +{ + std::vector input = json::to_ubjson(json({{"a", {1, 2}}})); + input.pop_back(); + + json _; + CHECK_THROWS_AS(_ = json::from_ubjson(input.begin(), input.end()), json::parse_error&); + CHECK(json::from_ubjson(input, true, false).is_discarded()); + CHECK(json::from_ubjson(input.begin(), input.end(), true, false).is_discarded()); + CHECK(json::from_ubjson(input.data(), input.size(), true, false).is_discarded()); + CHECK(json::from_ubjson({input.data(), input.size()}, true, false).is_discarded()); +} + +TEST_CASE("UBJSON SAX parsing stops at every event") +{ + // Containers are opened and closed by the loop that reads them; a SAX + // handler that rejects any event - including the end of a nested + // container - must stop the parse right there. + const auto count_events = [](const std::vector& input) + { + int events = 0; + while (true) + { + SaxCountdown scp(events); + if (json::sax_parse(input, &scp, json::input_format_t::ubjson)) + { + return events; + } + ++events; + REQUIRE(events < 1000); + } + }; + + // 20 events: every container kind closes inside another one + const json j = json::parse(R"({"a": [1, {"b": []}], "c": {"d": [[2]]}})"); + CHECK(count_events(json::to_ubjson(j)) == 20); + CHECK(count_events(json::to_ubjson(j, true)) == 20); + CHECK(count_events(json::to_ubjson(j, true, true)) == 20); +} + TEST_CASE("UBJSON optimized arrays of a valueless type are bounded") { // An element of type 'Z', 'T' or 'F' is encoded by its marker alone, so an diff --git a/tests/src/unit-wstring.cpp b/tests/src/unit-wstring.cpp index a38df3aaa..6e0aff29e 100644 --- a/tests/src/unit-wstring.cpp +++ b/tests/src/unit-wstring.cpp @@ -70,6 +70,8 @@ TEST_CASE("wide strings") CHECK_THROWS_WITH_AS(_ = json::parse(std::wstring{L'"', static_cast(0xDC00), L'"'}), error_low_surrogate, json::parse_error&); // a high surrogate followed by a non-low-surrogate unit is invalid CHECK_THROWS_WITH_AS(_ = json::parse(std::wstring{L'"', static_cast(0xD800), L'a', L'"'}), error_high_surrogate, json::parse_error&); + // ... also when the unit is above the low surrogates + CHECK_THROWS_WITH_AS(_ = json::parse(std::wstring{L'"', static_cast(0xD800), static_cast(0xE000), L'"'}), error_high_surrogate, json::parse_error&); // a lone low surrogate must not swallow the following unit: pairing // it with any second unit would produce valid UTF-8, so the error // has to report an ill-formed byte at the surrogate's own position @@ -99,6 +101,8 @@ TEST_CASE("wide strings") CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xDC00, u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"'", json::parse_error&); // a high surrogate followed by a non-low-surrogate unit is invalid CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xD800, u'a', u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"'", json::parse_error&); + // ... also when the unit is above the low surrogates + CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xD800, 0xE000, u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"'", json::parse_error&); // a lone low surrogate must not swallow the following unit: pairing // it with any second unit would produce valid UTF-8, so the error // has to report an ill-formed byte at the surrogate's own position From 6178982b8d37f1ec3c0954baf26a69c826a79425 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Sun, 27 Sep 2026 14:21:58 +0200 Subject: [PATCH 58/64] Compare unordered objects by key below the nesting bound (#5582) * Compare unordered objects by key below the nesting bound Values nested deeper than the nesting bound are compared without the call stack, walking both objects entry by entry. Two equal objects of a type that enumerates its entries in no fixed order - std::unordered_map, say - can be walked in different orders, so they compared unequal, and a deep copy compared unequal to its original. std::unordered_map's own operator== does not depend on the order, which is what applies above the bound. Where the keys differ, equality now finds the entry by its key instead. An ordering, and ordered_map, whose operator== compares its entries in sequence, still decide by the key. Signed-off-by: Niels Lohmann * Test unordered object equality without std::unordered_map basic_json instantiates std::pair while basic_json is still incomplete. The standard does not require std::unordered_map to support that, and libstdc++ 6 to 9 as well as the EDG front ends of icpc and nvc++ reject it, which broke the build of unit-comparison on those CI jobs. The test now uses an object type derived from std::map (which, as the default object type, works everywhere) whose comparator orders keys ascending or descending as chosen at construction, and whose operator== does not depend on the order of the entries - the property of std::unordered_map the test is about. Signed-off-by: Niels Lohmann * Compare the test object type's entries with std::all_of clang-tidy (readability-use-anyofallof) asked for std::all_of instead of the loop in unordered_object_t's operator==. The entry type is spelled out, as C++11 needs typename for base_type::value_type and C++20 reports it as redundant. Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- include/nlohmann/json.hpp | 25 ++++-- single_include/nlohmann/json.hpp | 25 ++++-- tests/src/unit-comparison.cpp | 127 +++++++++++++++++++++++++++++++ 3 files changed, 167 insertions(+), 10 deletions(-) diff --git a/include/nlohmann/json.hpp b/include/nlohmann/json.hpp index d18baf599..abd461d6c 100644 --- a/include/nlohmann/json.hpp +++ b/include/nlohmann/json.hpp @@ -1492,13 +1492,28 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec compare_keys(current.lhs_object_it->first, current.rhs_object_it->first, std::integral_constant {}); - if (key_result != compare_result::equal) - { - return key_result; - } - left = &(current.lhs_object_it->second); right = &(current.rhs_object_it->second); + + if (key_result != compare_result::equal) + { + // An object type without a fixed order of its entries - + // std::unordered_map, say - may enumerate two equal + // objects differently, and its operator== does not care. + // Equality then finds the entry by its key; an ordering, + // or an object type that compares its entries in + // sequence (ordered_map), is decided by the key itself. + const auto* rhs_object = current.rhs_value->m_data.m_value.object; + const auto found = (!Ordered && !detail::is_ordered_map::value) + ? rhs_object->find(current.lhs_object_it->first) + : rhs_object->cend(); + if (found == rhs_object->cend()) + { + return key_result; + } + right = &(found->second); + } + ++current.lhs_object_it; ++current.rhs_object_it; } diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index b5938be0e..ba41273de 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -26361,13 +26361,28 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec compare_keys(current.lhs_object_it->first, current.rhs_object_it->first, std::integral_constant {}); - if (key_result != compare_result::equal) - { - return key_result; - } - left = &(current.lhs_object_it->second); right = &(current.rhs_object_it->second); + + if (key_result != compare_result::equal) + { + // An object type without a fixed order of its entries - + // std::unordered_map, say - may enumerate two equal + // objects differently, and its operator== does not care. + // Equality then finds the entry by its key; an ordering, + // or an object type that compares its entries in + // sequence (ordered_map), is decided by the key itself. + const auto* rhs_object = current.rhs_value->m_data.m_value.object; + const auto found = (!Ordered && !detail::is_ordered_map::value) + ? rhs_object->find(current.lhs_object_it->first) + : rhs_object->cend(); + if (found == rhs_object->cend()) + { + return key_result; + } + right = &(found->second); + } + ++current.lhs_object_it; ++current.rhs_object_it; } diff --git a/tests/src/unit-comparison.cpp b/tests/src/unit-comparison.cpp index 9a6606256..69c0103c9 100644 --- a/tests/src/unit-comparison.cpp +++ b/tests/src/unit-comparison.cpp @@ -15,7 +15,13 @@ #include "doctest_compatibility.h" +#include + #include +#include +#include +#include +#include #define JSON_TESTS_PRIVATE #include @@ -745,6 +751,127 @@ TEST_CASE("regression #3868 - heterogeneous comparisons compile under C++20 (P24 } #endif +namespace +{ +// orders keys ascending or descending, as chosen when a map is created +template +class directed_less +{ + public: + directed_less() = default; + + explicit directed_less(const bool descending) noexcept + : m_descending(descending) + {} + + bool operator()(const Key& lhs, const Key& rhs) const + { + return m_descending ? rhs < lhs : lhs < rhs; + } + + private: + bool m_descending = false; +}; + +// An object type that, like std::unordered_map, enumerates its entries in no +// fixed order - ascending or descending by key, depending on how the map was +// created - and whose operator== does not depend on that order. +// std::unordered_map itself cannot be used here: the standard does not +// require it to accept an incomplete mapped type such as basic_json, and +// libstdc++ 6 to 9 as well as the EDG front ends of icpc and nvc++ reject +// basic_json. std::map, the default object type, works +// with all supported compilers. +template +struct unordered_object_t : std::map, Allocator> +{ + using base_type = std::map, Allocator>; + using base_type::base_type; + + friend bool operator==(const unordered_object_t& lhs, const unordered_object_t& rhs) + { + return lhs.size() == rhs.size() && std::all_of(lhs.begin(), lhs.end(), [&rhs](const std::pair& entry) + { + const auto it = rhs.find(entry.first); + return it != rhs.end() && it->second == entry.second; + }); + } + + friend bool operator!=(const unordered_object_t& lhs, const unordered_object_t& rhs) + { + return !(lhs == rhs); + } +}; +using unordered_json = nlohmann::basic_json; + +// the entries "0" to "9", enumerated in ascending or in descending order +unordered_json make_unordered_object(const bool descending) +{ + unordered_json j = unordered_json::object_t(directed_less(descending)); + for (int i = 0; i < 10; ++i) + { + j[std::to_string(i)] = i; + } + return j; +} + +template +Json nest(Json j, const std::size_t depth) +{ + for (std::size_t i = 0; i < depth; ++i) + { + Json outer = Json::object(); + outer["x"] = std::move(j); + j = std::move(outer); + } + return j; +} +} // namespace + +TEST_CASE("equality of objects whose entries have no fixed order") +{ + // Values nested deeper than a bound are compared without the call stack, + // entry by entry. That must agree with the object type's own operator==, + // which for unordered_object_t (as for std::unordered_map) does not + // depend on the order of the entries, and for ordered_map does. + REQUIRE(make_unordered_object(true).begin().key() == "9"); + REQUIRE(make_unordered_object(false).begin().key() == "0"); + + for (const std::size_t depth : std::vector {0, 200}) + { + CAPTURE(depth); + + const unordered_json descending = nest(make_unordered_object(true), depth); + const unordered_json ascending = nest(make_unordered_object(false), depth); + CHECK(descending == ascending); + CHECK_FALSE(descending != ascending); + + // a copy is equal to its original + const unordered_json copy = descending; // NOLINT(performance-unnecessary-copy-initialization) + CHECK(copy == descending); + + // a different value, a different key, or another entry still count + unordered_json other_value = make_unordered_object(true); + other_value["5"] = 42; + CHECK_FALSE(nest(other_value, depth) == ascending); + + unordered_json other_key = make_unordered_object(true); + other_key.erase("5"); + other_key["50"] = 5; + CHECK_FALSE(nest(other_key, depth) == ascending); + + unordered_json more_entries = make_unordered_object(true); + more_entries["10"] = 10; + CHECK_FALSE(nest(more_entries, depth) == ascending); + CHECK_FALSE(ascending == nest(more_entries, depth)); + + // ordered_json compares its entries in sequence + const nlohmann::ordered_json ab = nest(nlohmann::ordered_json({{"a", 1}, {"b", 2}}), depth); + const nlohmann::ordered_json ba = nest(nlohmann::ordered_json({{"b", 2}, {"a", 1}}), depth); + CHECK_FALSE(ab == ba); + CHECK(ab != ba); + } +} + TEST_CASE("containers are compared element by element") { // Containers nested deeper than a bound are compared without the call From f7972970a4d620a9807bc63217cde94db8ff0da9 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Sun, 27 Sep 2026 14:28:16 +0200 Subject: [PATCH 59/64] Throw instead of writing MessagePack lengths beyond UINT32_MAX (#5584) * Throw instead of writing MessagePack lengths beyond UINT32_MAX MessagePack stores the length of a string, binary value, array, or object in at most 32 bits. For a larger value, to_msgpack wrote no length at all, so the output could not be read back. It now throws out_of_range.412, which BSON already uses for its 32-bit length fields. The check lives in one function, so each length is written by an if/else chain that ends in a plain else, without a condition that can never be false. It is tested with string and binary types that report a size beyond UINT32_MAX without allocating it, like the BSON tests do. Signed-off-by: Niels Lohmann * Fix the CI failures of the MessagePack length check - mark to_msgpack_length's value as used when exceptions are disabled (-Wunused-parameter, misc-unused-parameters) - put "Exception safety" before "Exceptions" in to_msgpack.md, as the documentation style check requires - create the test's string value from its type: constructing it from a beyond_uint32_string_t considers the std::filesystem::path conversion, which libstdc++ 10 reports as ambiguous for a class derived from std::string (clang 13) Signed-off-by: Niels Lohmann * Skip the MessagePack string length test for clang with libstdc++ 10 C++17 builds consider the std::filesystem::path conversion for the string type, and with clang and libstdc++ 10 that conversion is ambiguous for a class derived from std::string. Creating the value from its type did not avoid it, since any basic_json with that string type instantiates the check. The binary and ext cases are still tested there. Signed-off-by: Niels Lohmann * Keep the MessagePack string test type and its alias in one block astyle indented the alias oddly when it had an #ifdef of its own after the binary alias; declare it right after the string type, in the same block. Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- docs/mkdocs/docs/api/basic_json/to_msgpack.md | 10 +++ .../features/binary_formats/messagepack.md | 2 + docs/mkdocs/docs/home/exceptions.md | 14 +++- .../nlohmann/detail/output/binary_writer.hpp | 53 ++++++------ single_include/nlohmann/json.hpp | 53 ++++++------ tests/src/unit-msgpack.cpp | 84 ++++++++++++++++++- 6 files changed, 152 insertions(+), 64 deletions(-) diff --git a/docs/mkdocs/docs/api/basic_json/to_msgpack.md b/docs/mkdocs/docs/api/basic_json/to_msgpack.md index 66b104f52..b3bcaab7f 100644 --- a/docs/mkdocs/docs/api/basic_json/to_msgpack.md +++ b/docs/mkdocs/docs/api/basic_json/to_msgpack.md @@ -34,6 +34,15 @@ The exact mapping and its limitations are described on a [dedicated page](../../ Strong guarantee: if an exception is thrown, there are no changes in the JSON value. +## Exceptions + +- Throws [`out_of_range.412`](../../home/exceptions.md#jsonexceptionout_of_range412) if the length of a string, binary + value, array, or object exceeds 4294967295, the maximum MessagePack can store; example: + `"MessagePack length 4294967296 exceeds maximum of 4294967295"` +- Throws [`out_of_range.415`](../../home/exceptions.md#jsonexceptionout_of_range415) if the subtype of a binary value + exceeds 255, the maximum of the MessagePack ext type; example: + `"subtype 70000 is too large for the MessagePack ext type (max 255)"` + ## Complexity Linear in the size of the JSON value `j`. @@ -65,3 +74,4 @@ Linear in the size of the JSON value `j`. ## Version history - Added in version 2.0.9. +- Throws `out_of_range.412` and `out_of_range.415` since version 3.13.0. diff --git a/docs/mkdocs/docs/features/binary_formats/messagepack.md b/docs/mkdocs/docs/features/binary_formats/messagepack.md index a434909c4..0ca82c145 100644 --- a/docs/mkdocs/docs/features/binary_formats/messagepack.md +++ b/docs/mkdocs/docs/features/binary_formats/messagepack.md @@ -65,6 +65,8 @@ specification: - arrays with more than 4294967295 elements - objects with more than 4294967295 elements + Serializing such a value throws [`out_of_range.412`](../../home/exceptions.md#jsonexceptionout_of_range412). + !!! info "NaN/infinity handling" `NaN`, `Infinity`, and `-Infinity` are serialized as a MessagePack float 32 (type 0xCA, 5 bytes total), diff --git a/docs/mkdocs/docs/home/exceptions.md b/docs/mkdocs/docs/home/exceptions.md index ee76596f6..bf18baab1 100644 --- a/docs/mkdocs/docs/home/exceptions.md +++ b/docs/mkdocs/docs/home/exceptions.md @@ -932,19 +932,25 @@ A JSON Patch `add` operation cannot be applied because the target location's par ### json.exception.out_of_range.412 -BSON stores the length of documents, arrays, strings, and binary values in a signed 32-bit integer. This exception is thrown when a value is too large to be described by such a length field. +BSON stores the length of documents, arrays, strings, and binary values in a signed 32-bit integer, and MessagePack +stores the length of strings, binary values, arrays, and objects in at most an unsigned 32-bit integer. This exception +is thrown when a value is too large to be described by such a length field. -!!! failure "Example message" +!!! failure "Example messages" ``` BSON length 2147483661 exceeds maximum of 2147483647 ``` + ``` + MessagePack length 4294967296 exceeds maximum of 4294967295 + ``` !!! note - This exception was added in version 3.13.0. Before that, the length was silently truncated, and + This exception was added in version 3.13.0. Before that, the BSON length was silently truncated, and [`to_bson`](../api/basic_json/to_bson.md) produced documents with negative length prefixes that - [`from_bson`](../api/basic_json/from_bson.md) rejected. + [`from_bson`](../api/basic_json/from_bson.md) rejected; [`to_msgpack`](../api/basic_json/to_msgpack.md) wrote such + a value without any length, producing output that could not be read back. ### json.exception.out_of_range.413 diff --git a/include/nlohmann/detail/output/binary_writer.hpp b/include/nlohmann/detail/output/binary_writer.hpp index 9c41e2962..5b11a3b2f 100644 --- a/include/nlohmann/detail/output/binary_writer.hpp +++ b/include/nlohmann/detail/output/binary_writer.hpp @@ -291,6 +291,23 @@ class binary_writer } } + /*! + @brief check that @a length fits into the 32 bits that MessagePack stores + the length of a string, binary value, array, or object in + @return the length as an unsigned 32-bit integer + @throw out_of_range.412 if @a length exceeds the range of std::uint32_t + */ + static std::uint32_t to_msgpack_length(const std::size_t length, const BasicJsonType& j) + { + if (JSON_HEDLEY_UNLIKELY(!value_in_range_of(length))) + { + JSON_THROW(out_of_range::create(412, concat("MessagePack length ", std::to_string(length), " exceeds maximum of ", std::to_string((std::numeric_limits::max)())), &j)); + } + + static_cast(j); + return static_cast(length); + } + /*! @param[in] j JSON value to serialize */ @@ -430,7 +447,7 @@ class binary_writer case value_t::string: { // step 1: write control byte and the string length - const auto N = j.m_data.m_value.string->size(); + const auto N = to_msgpack_length(j.m_data.m_value.string->size(), j); if (N <= 31) { // fixstr @@ -448,17 +465,12 @@ class binary_writer oa.write_character(to_char_type(0xDA)); write_number(static_cast(N)); } - else if (N <= (std::numeric_limits::max)()) + else { // str 32 oa.write_character(to_char_type(0xDB)); write_number(static_cast(N)); } - else - { - JSON_THROW(out_of_range::create(412, concat("MessagePack size ", std::to_string(N), " exceeds maximum of ", - std::to_string((std::numeric_limits::max)())), &j)); - } // step 2: write the string oa.write_characters( @@ -470,7 +482,7 @@ class binary_writer case value_t::array: { // step 1: write control byte and the array size - const auto N = j.m_data.m_value.array->size(); + const auto N = to_msgpack_length(j.m_data.m_value.array->size(), j); if (N <= 15) { // fixarray @@ -482,17 +494,12 @@ class binary_writer oa.write_character(to_char_type(0xDC)); write_number(static_cast(N)); } - else if (N <= (std::numeric_limits::max)()) + else { // array 32 oa.write_character(to_char_type(0xDD)); write_number(static_cast(N)); } - else - { - JSON_THROW(out_of_range::create(412, concat("MessagePack size ", std::to_string(N), " exceeds maximum of ", - std::to_string((std::numeric_limits::max)())), &j)); - } // step 2: write each element for (const auto& el : *j.m_data.m_value.array) @@ -509,7 +516,7 @@ class binary_writer const bool use_ext = j.m_data.m_value.binary->has_subtype(); // step 1: write control byte and the byte string length - const auto N = j.m_data.m_value.binary->size(); + const auto N = to_msgpack_length(j.m_data.m_value.binary->size(), j); if (N <= (std::numeric_limits::max)()) { std::uint8_t output_type{}; @@ -561,7 +568,7 @@ class binary_writer oa.write_character(to_char_type(output_type)); write_number(static_cast(N)); } - else if (N <= (std::numeric_limits::max)()) + else { const std::uint8_t output_type = use_ext ? 0xC9 // ext 32 @@ -570,11 +577,6 @@ class binary_writer oa.write_character(to_char_type(output_type)); write_number(static_cast(N)); } - else - { - JSON_THROW(out_of_range::create(412, concat("MessagePack size ", std::to_string(N), " exceeds maximum of ", - std::to_string((std::numeric_limits::max)())), &j)); - } // step 1.5: if this is an ext type, write the subtype if (use_ext) @@ -598,7 +600,7 @@ class binary_writer case value_t::object: { // step 1: write control byte and the object size - const auto N = j.m_data.m_value.object->size(); + const auto N = to_msgpack_length(j.m_data.m_value.object->size(), j); if (N <= 15) { // fixmap @@ -610,17 +612,12 @@ class binary_writer oa.write_character(to_char_type(0xDE)); write_number(static_cast(N)); } - else if (N <= (std::numeric_limits::max)()) + else { // map 32 oa.write_character(to_char_type(0xDF)); write_number(static_cast(N)); } - else - { - JSON_THROW(out_of_range::create(412, concat("MessagePack size ", std::to_string(N), " exceeds maximum of ", - std::to_string((std::numeric_limits::max)())), &j)); - } // step 2: write each element for (const auto& el : *j.m_data.m_value.object) diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index ba41273de..bf1aee332 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -19756,6 +19756,23 @@ class binary_writer } } + /*! + @brief check that @a length fits into the 32 bits that MessagePack stores + the length of a string, binary value, array, or object in + @return the length as an unsigned 32-bit integer + @throw out_of_range.412 if @a length exceeds the range of std::uint32_t + */ + static std::uint32_t to_msgpack_length(const std::size_t length, const BasicJsonType& j) + { + if (JSON_HEDLEY_UNLIKELY(!value_in_range_of(length))) + { + JSON_THROW(out_of_range::create(412, concat("MessagePack length ", std::to_string(length), " exceeds maximum of ", std::to_string((std::numeric_limits::max)())), &j)); + } + + static_cast(j); + return static_cast(length); + } + /*! @param[in] j JSON value to serialize */ @@ -19895,7 +19912,7 @@ class binary_writer case value_t::string: { // step 1: write control byte and the string length - const auto N = j.m_data.m_value.string->size(); + const auto N = to_msgpack_length(j.m_data.m_value.string->size(), j); if (N <= 31) { // fixstr @@ -19913,17 +19930,12 @@ class binary_writer oa.write_character(to_char_type(0xDA)); write_number(static_cast(N)); } - else if (N <= (std::numeric_limits::max)()) + else { // str 32 oa.write_character(to_char_type(0xDB)); write_number(static_cast(N)); } - else - { - JSON_THROW(out_of_range::create(412, concat("MessagePack size ", std::to_string(N), " exceeds maximum of ", - std::to_string((std::numeric_limits::max)())), &j)); - } // step 2: write the string oa.write_characters( @@ -19935,7 +19947,7 @@ class binary_writer case value_t::array: { // step 1: write control byte and the array size - const auto N = j.m_data.m_value.array->size(); + const auto N = to_msgpack_length(j.m_data.m_value.array->size(), j); if (N <= 15) { // fixarray @@ -19947,17 +19959,12 @@ class binary_writer oa.write_character(to_char_type(0xDC)); write_number(static_cast(N)); } - else if (N <= (std::numeric_limits::max)()) + else { // array 32 oa.write_character(to_char_type(0xDD)); write_number(static_cast(N)); } - else - { - JSON_THROW(out_of_range::create(412, concat("MessagePack size ", std::to_string(N), " exceeds maximum of ", - std::to_string((std::numeric_limits::max)())), &j)); - } // step 2: write each element for (const auto& el : *j.m_data.m_value.array) @@ -19974,7 +19981,7 @@ class binary_writer const bool use_ext = j.m_data.m_value.binary->has_subtype(); // step 1: write control byte and the byte string length - const auto N = j.m_data.m_value.binary->size(); + const auto N = to_msgpack_length(j.m_data.m_value.binary->size(), j); if (N <= (std::numeric_limits::max)()) { std::uint8_t output_type{}; @@ -20026,7 +20033,7 @@ class binary_writer oa.write_character(to_char_type(output_type)); write_number(static_cast(N)); } - else if (N <= (std::numeric_limits::max)()) + else { const std::uint8_t output_type = use_ext ? 0xC9 // ext 32 @@ -20035,11 +20042,6 @@ class binary_writer oa.write_character(to_char_type(output_type)); write_number(static_cast(N)); } - else - { - JSON_THROW(out_of_range::create(412, concat("MessagePack size ", std::to_string(N), " exceeds maximum of ", - std::to_string((std::numeric_limits::max)())), &j)); - } // step 1.5: if this is an ext type, write the subtype if (use_ext) @@ -20063,7 +20065,7 @@ class binary_writer case value_t::object: { // step 1: write control byte and the object size - const auto N = j.m_data.m_value.object->size(); + const auto N = to_msgpack_length(j.m_data.m_value.object->size(), j); if (N <= 15) { // fixmap @@ -20075,17 +20077,12 @@ class binary_writer oa.write_character(to_char_type(0xDE)); write_number(static_cast(N)); } - else if (N <= (std::numeric_limits::max)()) + else { // map 32 oa.write_character(to_char_type(0xDF)); write_number(static_cast(N)); } - else - { - JSON_THROW(out_of_range::create(412, concat("MessagePack size ", std::to_string(N), " exceeds maximum of ", - std::to_string((std::numeric_limits::max)())), &j)); - } // step 2: write each element for (const auto& el : *j.m_data.m_value.object) diff --git a/tests/src/unit-msgpack.cpp b/tests/src/unit-msgpack.cpp index 0b8c6ca63..d86e6f068 100644 --- a/tests/src/unit-msgpack.cpp +++ b/tests/src/unit-msgpack.cpp @@ -14,6 +14,7 @@ using nlohmann::json; using namespace nlohmann::literals; // NOLINT(google-build-using-namespace) #endif +#include // SIZE_MAX, UINT32_MAX #include #include #include @@ -2226,7 +2227,7 @@ TEST_CASE("MessagePack Size above uint32 for array") CHECK_THROWS_WITH_AS( huge_array_json::to_msgpack(j), - "[json.exception.out_of_range.412] MessagePack size 4294967296 exceeds maximum of 4294967295", + "[json.exception.out_of_range.412] MessagePack length 4294967296 exceeds maximum of 4294967295", json::out_of_range&); array.fake_size = false; @@ -2279,7 +2280,7 @@ TEST_CASE("MessagePack Size above uint32 for object") CHECK_THROWS_WITH_AS( huge_object_json::to_msgpack(j), - "[json.exception.out_of_range.412] MessagePack size 4294967296 exceeds maximum of 4294967295", + "[json.exception.out_of_range.412] MessagePack length 4294967296 exceeds maximum of 4294967295", json::out_of_range&); object.fake_size = false; @@ -2315,7 +2316,7 @@ TEST_CASE("MessagePack Size above uint32 for string") CHECK_THROWS_WITH_AS( huge_string_json::to_msgpack(j), - "[json.exception.out_of_range.412] MessagePack size 4294967296 exceeds maximum of 4294967295", + "[json.exception.out_of_range.412] MessagePack length 4294967296 exceeds maximum of 4294967295", json::out_of_range&); } @@ -2352,7 +2353,82 @@ TEST_CASE("MessagePack Size above uint32 for binary") CHECK_THROWS_WITH_AS( huge_binary_json::to_msgpack(j), - "[json.exception.out_of_range.412] MessagePack size 4294967296 exceeds maximum of 4294967295", + "[json.exception.out_of_range.412] MessagePack length 4294967296 exceeds maximum of 4294967295", json::out_of_range&); } +namespace +{ +// types that report a size beyond UINT32_MAX without allocating that much +// memory, so the MessagePack length limit can be tested cheaply; see the +// similar types in unit-bson.cpp +std::size_t beyond_uint32_size() +{ + return static_cast((std::numeric_limits::max)()) + 1; +} + +class beyond_uint32_binary_t : public std::vector +{ + public: + using std::vector::vector; + + size_type size() const noexcept // NOLINT(readability-convert-member-functions-to-static) + { + return beyond_uint32_size(); + } +}; + +// with clang and libstdc++ 10, the std::filesystem::path conversion that +// C++17 builds consider for every string type is ambiguous for a class +// derived from std::string, so the string case is not tested there +#if !(defined(__clang__) && defined(_GLIBCXX_RELEASE) && _GLIBCXX_RELEASE < 11) + #define JSON_TEST_BEYOND_UINT32_STRING 1 +#endif + +#ifdef JSON_TEST_BEYOND_UINT32_STRING +class beyond_uint32_string_t : public std::string +{ + public: + using std::string::string; + + size_type size() const noexcept // NOLINT(readability-convert-member-functions-to-static) + { + return beyond_uint32_size(); + } +}; + +using beyond_uint32_string_json = nlohmann::basic_json < + std::map, std::vector, beyond_uint32_string_t, bool, std::int64_t, std::uint64_t, + double, std::allocator, nlohmann::adl_serializer, std::vector, void >; +#endif + +using beyond_uint32_binary_json = nlohmann::basic_json < + std::map, std::vector, std::string, bool, std::int64_t, std::uint64_t, + double, std::allocator, nlohmann::adl_serializer, beyond_uint32_binary_t, void >; +} // namespace + +TEST_CASE("MessagePack lengths beyond UINT32_MAX cannot be serialized") +{ + // MessagePack stores the length of a string, binary value, array, or + // object in at most 32 bits; a larger one used to be written without any + // length at all +#if SIZE_MAX > UINT32_MAX + { + const char* const expected = "[json.exception.out_of_range.412] MessagePack length 4294967296 exceeds maximum of 4294967295"; + + const beyond_uint32_binary_json binary = beyond_uint32_binary_json::binary(beyond_uint32_binary_t{}); + CHECK_THROWS_WITH_AS(beyond_uint32_binary_json::to_msgpack(binary), expected, beyond_uint32_binary_json::out_of_range&); + + const beyond_uint32_binary_json ext = beyond_uint32_binary_json::binary(beyond_uint32_binary_t{}, 42); + CHECK_THROWS_WITH_AS(beyond_uint32_binary_json::to_msgpack(ext), expected, beyond_uint32_binary_json::out_of_range&); + +#ifdef JSON_TEST_BEYOND_UINT32_STRING + // created from its type rather than from a beyond_uint32_string_t: + // that would consider the std::filesystem::path conversion, which + // libstdc++ 10 cannot decide for a class derived from std::string + const beyond_uint32_string_json string(beyond_uint32_string_json::value_t::string); + CHECK_THROWS_WITH_AS(beyond_uint32_string_json::to_msgpack(string), expected, beyond_uint32_string_json::out_of_range&); +#endif + } +#endif +} From 98e00d22e564de162532fb7a0da3d634af5c2bff Mon Sep 17 00:00:00 2001 From: bucketbase26 Date: Sun, 27 Sep 2026 17:58:38 +0530 Subject: [PATCH 60/64] Cut test suite runtime in binary roundtrips and integer sweeps (#5519) * Cut test suite runtime in binary roundtrips and integer sweeps The Linux CI jobs pass --no-skip, so skip() does not help there. Parse each corpus file once in the binary roundtrip loops instead of four times. Sample the 16-bit integer ranges with stride 7 (still hits every low byte) and always keep the endpoints. Also drop the 5M-node parse test to 500k, which still covers the non-recursive destructor, and move jeopardy.json into its own skipped test so the cheaper binary-format size checks actually run. See #5418. Signed-off-by: ayush-singh-0601 * Drop useless int32_t casts in the sampled integer loops ci_test_gcc compiles with -Werror=useless-cast. On that compiler int32_t is int, so static_cast of the loop bound is an error. The bounds are already int, and the sampled values do not change. Signed-off-by: ayush-singh-0601 * Revert unit-binary_formats.cpp to develop and fix comment Revert tests/src/unit-binary_formats.cpp to its develop state. The test-case split made valgrind jobs slower instead of faster, because the cheaper corpus files (canada/twitter/citm/sample) now ran under valgrind where they never did before. Fix the next_integer_sample comment: the function has no 'first' parameter, so describe what the function actually does. Signed-off-by: ayush-singh-0601 --------- Signed-off-by: ayush-singh-0601 --- tests/src/test_utils.hpp | 18 +++++++++++++++ tests/src/unit-bjdata.cpp | 32 ++++++-------------------- tests/src/unit-cbor.cpp | 42 +++++++---------------------------- tests/src/unit-large_json.cpp | 2 +- tests/src/unit-msgpack.cpp | 40 ++++++--------------------------- tests/src/unit-ubjson.cpp | 40 ++++++--------------------------- 6 files changed, 48 insertions(+), 126 deletions(-) diff --git a/tests/src/test_utils.hpp b/tests/src/test_utils.hpp index 4c81a8ef4..b2a381fb0 100644 --- a/tests/src/test_utils.hpp +++ b/tests/src/test_utils.hpp @@ -9,6 +9,7 @@ #pragma once #include // uint8_t +#include // size_t #include // ifstream, istreambuf_iterator, ios #include // vector @@ -24,6 +25,23 @@ namespace utils template inline void ignore_return_value(T&& /*unused*/) noexcept {} +// Advance i toward last (inclusive) by stride, always visiting last. +// stride 7 is coprime to 256, so every low-byte residue is still hit. +template +T next_integer_sample(T i, T last, T stride) +{ + if (i >= last) + { + return static_cast(last + 1); + } + if (stride > 0 && i > static_cast(last - stride)) + { + return last; + } + const T n = static_cast(i + stride); + return n < last ? n : last; +} + inline std::vector read_binary_file(const std::string& filename) { std::ifstream file(filename, std::ios::binary); diff --git a/tests/src/unit-bjdata.cpp b/tests/src/unit-bjdata.cpp index 03c69431b..a58507c15 100644 --- a/tests/src/unit-bjdata.cpp +++ b/tests/src/unit-bjdata.cpp @@ -418,7 +418,7 @@ TEST_CASE("BJData") SECTION("-32768..-129 (int16)") { - for (int32_t i = -32768; i <= -129; ++i) + for (int32_t i = -32768; i <= -129; i = utils::next_integer_sample(i, -129, 7)) { CAPTURE(i) @@ -578,7 +578,7 @@ TEST_CASE("BJData") SECTION("256..32767 (int16)") { - for (size_t i = 256; i <= 32767; ++i) + for (size_t i = 256; i <= 32767; i = utils::next_integer_sample(i, static_cast(32767), static_cast(7))) { CAPTURE(i) @@ -911,7 +911,7 @@ TEST_CASE("BJData") SECTION("256..32767 (int16)") { - for (size_t i = 256; i <= 32767; ++i) + for (size_t i = 256; i <= 32767; i = utils::next_integer_sample(i, static_cast(32767), static_cast(7))) { CAPTURE(i) @@ -4541,45 +4541,27 @@ TEST_CASE("BJData roundtrips" * doctest::skip()) { CAPTURE(filename) + std::ifstream f_json(filename); + const json j1 = json::parse(f_json); + auto packed = utils::read_binary_file(filename + ".bjdata"); + { INFO_WITH_TEMP(filename + ": std::vector"); - // parse JSON file - std::ifstream f_json(filename); - const json j1 = json::parse(f_json); - - // parse BJData file - auto packed = utils::read_binary_file(filename + ".bjdata"); json j2; CHECK_NOTHROW(j2 = json::from_bjdata(packed)); - - // compare parsed JSON values CHECK(j1 == j2); } { INFO_WITH_TEMP(filename + ": std::ifstream"); - // parse JSON file - std::ifstream f_json(filename); - const json j1 = json::parse(f_json); - - // parse BJData file std::ifstream f_bjdata(filename + ".bjdata", std::ios::binary); json j2; CHECK_NOTHROW(j2 = json::from_bjdata(f_bjdata)); - - // compare parsed JSON values CHECK(j1 == j2); } { INFO_WITH_TEMP(filename + ": output to output adapters"); - // parse JSON file - std::ifstream f_json(filename); - json const j1 = json::parse(f_json); - - // parse BJData file - auto packed = utils::read_binary_file(filename + ".bjdata"); - { INFO_WITH_TEMP(filename + ": output adapters: std::vector"); std::vector vec; diff --git a/tests/src/unit-cbor.cpp b/tests/src/unit-cbor.cpp index 353281d5e..77f9a10e6 100644 --- a/tests/src/unit-cbor.cpp +++ b/tests/src/unit-cbor.cpp @@ -291,7 +291,7 @@ TEST_CASE("CBOR") SECTION("-65536..-257") { - for (int32_t i = -65536; i <= -257; ++i) + for (int32_t i = -65536; i <= -257; i = utils::next_integer_sample(i, -257, 7)) { CAPTURE(i) @@ -479,7 +479,7 @@ TEST_CASE("CBOR") SECTION("256..65535") { - for (size_t i = 256; i <= 65535; ++i) + for (size_t i = 256; i <= 65535; i = utils::next_integer_sample(i, static_cast(65535), static_cast(7))) { CAPTURE(i) @@ -614,7 +614,7 @@ TEST_CASE("CBOR") SECTION("-32768..-129 (int 16)") { - for (int16_t i = -32768; i <= static_cast(-129); ++i) + for (int16_t i = -32768; i <= static_cast(-129); i = utils::next_integer_sample(i, static_cast(-129), static_cast(7))) { CAPTURE(i) @@ -719,7 +719,7 @@ TEST_CASE("CBOR") SECTION("256..65535 (two-byte uint16_t)") { - for (size_t i = 256; i <= 65535; ++i) + for (size_t i = 256; i <= 65535; i = utils::next_integer_sample(i, static_cast(65535), static_cast(7))) { CAPTURE(i) @@ -2529,60 +2529,34 @@ TEST_CASE("CBOR roundtrips" * doctest::skip()) { CAPTURE(filename) + std::ifstream f_json(filename); + const json j1 = json::parse(f_json); + const auto packed = utils::read_binary_file(filename + ".cbor"); + { INFO_WITH_TEMP(filename + ": std::vector"); - // parse JSON file - std::ifstream f_json(filename); - const json j1 = json::parse(f_json); - - // parse CBOR file - const auto packed = utils::read_binary_file(filename + ".cbor"); json j2; CHECK_NOTHROW(j2 = json::from_cbor(packed)); - - // compare parsed JSON values CHECK(j1 == j2); } { INFO_WITH_TEMP(filename + ": std::ifstream"); - // parse JSON file - std::ifstream f_json(filename); - const json j1 = json::parse(f_json); - - // parse CBOR file std::ifstream f_cbor(filename + ".cbor", std::ios::binary); json j2; CHECK_NOTHROW(j2 = json::from_cbor(f_cbor)); - - // compare parsed JSON values CHECK(j1 == j2); } { INFO_WITH_TEMP(filename + ": uint8_t* and size"); - // parse JSON file - std::ifstream f_json(filename); - const json j1 = json::parse(f_json); - - // parse CBOR file - const auto packed = utils::read_binary_file(filename + ".cbor"); json j2; CHECK_NOTHROW(j2 = json::from_cbor({packed.data(), packed.size()})); - - // compare parsed JSON values CHECK(j1 == j2); } { INFO_WITH_TEMP(filename + ": output to output adapters"); - // parse JSON file - std::ifstream f_json(filename); - json const j1 = json::parse(f_json); - - // parse CBOR file - const auto packed = utils::read_binary_file(filename + ".cbor"); - if (exclude_packed.count(filename) == 0u) { { diff --git a/tests/src/unit-large_json.cpp b/tests/src/unit-large_json.cpp index 3734966d5..8204ed9b6 100644 --- a/tests/src/unit-large_json.cpp +++ b/tests/src/unit-large_json.cpp @@ -18,7 +18,7 @@ TEST_CASE("tests on very large JSONs") { SECTION("issue #1419 - Segmentation fault (stack overflow) due to unbounded recursion") { - const auto depth = 5000000; + const auto depth = 500000; std::string s(static_cast(2 * depth), '['); std::fill(s.begin() + depth, s.end(), ']'); diff --git a/tests/src/unit-msgpack.cpp b/tests/src/unit-msgpack.cpp index d86e6f068..cfb047495 100644 --- a/tests/src/unit-msgpack.cpp +++ b/tests/src/unit-msgpack.cpp @@ -256,7 +256,7 @@ TEST_CASE("MessagePack") SECTION("256..65535 (int 16)") { - for (size_t i = 256; i <= 65535; ++i) + for (size_t i = 256; i <= 65535; i = utils::next_integer_sample(i, static_cast(65535), static_cast(7))) { CAPTURE(i) @@ -441,7 +441,7 @@ TEST_CASE("MessagePack") SECTION("-32768..-129 (int 16)") { - for (int16_t i = -32768; i <= static_cast(-129); ++i) + for (int16_t i = -32768; i <= static_cast(-129); i = utils::next_integer_sample(i, static_cast(-129), static_cast(7))) { CAPTURE(i) @@ -647,7 +647,7 @@ TEST_CASE("MessagePack") SECTION("256..65535 (uint 16)") { - for (size_t i = 256; i <= 65535; ++i) + for (size_t i = 256; i <= 65535; i = utils::next_integer_sample(i, static_cast(65535), static_cast(7))) { CAPTURE(i) @@ -2043,60 +2043,34 @@ TEST_CASE("MessagePack roundtrips" * doctest::skip()) { CAPTURE(filename) + std::ifstream f_json(filename); + const json j1 = json::parse(f_json); + auto packed = utils::read_binary_file(filename + ".msgpack"); + { INFO_WITH_TEMP(filename + ": std::vector"); - // parse JSON file - std::ifstream f_json(filename); - const json j1 = json::parse(f_json); - - // parse MessagePack file - auto packed = utils::read_binary_file(filename + ".msgpack"); json j2; CHECK_NOTHROW(j2 = json::from_msgpack(packed)); - - // compare parsed JSON values CHECK(j1 == j2); } { INFO_WITH_TEMP(filename + ": std::ifstream"); - // parse JSON file - std::ifstream f_json(filename); - const json j1 = json::parse(f_json); - - // parse MessagePack file std::ifstream f_msgpack(filename + ".msgpack", std::ios::binary); json j2; CHECK_NOTHROW(j2 = json::from_msgpack(f_msgpack)); - - // compare parsed JSON values CHECK(j1 == j2); } { INFO_WITH_TEMP(filename + ": uint8_t* and size"); - // parse JSON file - std::ifstream f_json(filename); - const json j1 = json::parse(f_json); - - // parse MessagePack file - auto packed = utils::read_binary_file(filename + ".msgpack"); json j2; CHECK_NOTHROW(j2 = json::from_msgpack({packed.data(), packed.size()})); - - // compare parsed JSON values CHECK(j1 == j2); } { INFO_WITH_TEMP(filename + ": output to output adapters"); - // parse JSON file - std::ifstream f_json(filename); - json const j1 = json::parse(f_json); - - // parse MessagePack file - auto packed = utils::read_binary_file(filename + ".msgpack"); - if (exclude_packed.count(filename) == 0u) { { diff --git a/tests/src/unit-ubjson.cpp b/tests/src/unit-ubjson.cpp index c8a300e72..c315fac94 100644 --- a/tests/src/unit-ubjson.cpp +++ b/tests/src/unit-ubjson.cpp @@ -265,7 +265,7 @@ TEST_CASE("UBJSON") SECTION("-32768..-129 (int16)") { - for (int32_t i = -32768; i <= -129; ++i) + for (int32_t i = -32768; i <= -129; i = utils::next_integer_sample(i, -129, 7)) { CAPTURE(i) @@ -425,7 +425,7 @@ TEST_CASE("UBJSON") SECTION("256..32767 (int16)") { - for (size_t i = 256; i <= 32767; ++i) + for (size_t i = 256; i <= 32767; i = utils::next_integer_sample(i, static_cast(32767), static_cast(7))) { CAPTURE(i) @@ -631,7 +631,7 @@ TEST_CASE("UBJSON") SECTION("256..32767 (int16)") { - for (size_t i = 256; i <= 32767; ++i) + for (size_t i = 256; i <= 32767; i = utils::next_integer_sample(i, static_cast(32767), static_cast(7))) { CAPTURE(i) @@ -2980,60 +2980,34 @@ TEST_CASE("UBJSON roundtrips" * doctest::skip()) { CAPTURE(filename) + std::ifstream f_json(filename); + json const j1 = json::parse(f_json); + auto const packed = utils::read_binary_file(filename + ".ubjson"); + { INFO_WITH_TEMP(filename + ": std::vector"); - // parse JSON file - std::ifstream f_json(filename); - json const j1 = json::parse(f_json); - - // parse UBJSON file - auto const packed = utils::read_binary_file(filename + ".ubjson"); json j2; CHECK_NOTHROW(j2 = json::from_ubjson(packed)); - - // compare parsed JSON values CHECK(j1 == j2); } { INFO_WITH_TEMP(filename + ": std::ifstream"); - // parse JSON file - std::ifstream f_json(filename); - json const j1 = json::parse(f_json); - - // parse UBJSON file std::ifstream f_ubjson(filename + ".ubjson", std::ios::binary); json j2; CHECK_NOTHROW(j2 = json::from_ubjson(f_ubjson)); - - // compare parsed JSON values CHECK(j1 == j2); } { INFO_WITH_TEMP(filename + ": uint8_t* and size"); - // parse JSON file - std::ifstream f_json(filename); - const json j1 = json::parse(f_json); - - // parse UBJSON file - auto const packed = utils::read_binary_file(filename + ".ubjson"); json j2; CHECK_NOTHROW(j2 = json::from_ubjson({packed.data(), packed.size()})); - - // compare parsed JSON values CHECK(j1 == j2); } { INFO_WITH_TEMP(filename + ": output to output adapters"); - // parse JSON file - std::ifstream f_json(filename); - json const j1 = json::parse(f_json); - - // parse UBJSON file - auto const packed = utils::read_binary_file(filename + ".ubjson"); - { INFO_WITH_TEMP(filename + ": output adapters: std::vector"); std::vector vec; From 6bd106893a2b56296711f71fa2a80ecf9e9798cd Mon Sep 17 00:00:00 2001 From: Kartikey Negi <65110918+ReturnKartikey@users.noreply.github.com> Date: Sun, 27 Sep 2026 17:58:55 +0530 Subject: [PATCH 61/64] Fix CBOR tag handling in cbor_tag_handler_t::store for non-binary items (#5559) When using cbor_tag_handler_t::store, tags 0xD8-0xDB previously assumed that the tagged item was a byte string, unconditionally attempting to parse binary data and failing on valid CBOR documents containing tags applied to integers, strings, arrays, or objects (such as self-describe tag 55799). Check whether the tagged data item is a byte string (0x40-0x5B or 0x5F). If it is a byte string, store the subtype on the binary value as before. Otherwise, iteratively process the tagged value in the driver loop using item_read so that chained tags do not consume native stack space. Part of #5316. Signed-off-by: ReturnKartikey --- .../docs/api/basic_json/cbor_tag_handler_t.md | 2 +- .../docs/features/binary_formats/cbor.md | 2 +- .../nlohmann/detail/input/binary_reader.hpp | 25 ++++-- single_include/nlohmann/json.hpp | 25 ++++-- tests/src/unit-cbor.cpp | 85 +++++++++++++++++++ 5 files changed, 127 insertions(+), 12 deletions(-) diff --git a/docs/mkdocs/docs/api/basic_json/cbor_tag_handler_t.md b/docs/mkdocs/docs/api/basic_json/cbor_tag_handler_t.md index e19c3edd9..cea009e4e 100644 --- a/docs/mkdocs/docs/api/basic_json/cbor_tag_handler_t.md +++ b/docs/mkdocs/docs/api/basic_json/cbor_tag_handler_t.md @@ -18,7 +18,7 @@ ignore : ignore tags store -: store tagged values as binary container with subtype (for bytes 0xd8..0xdb) +: store tagged byte strings (for bytes 0xd8..0xdb) as binary values with the tag as subtype; other tagged values are read as if the tag were ignored. If several tags precede a byte string, only the innermost one is stored. ## Examples diff --git a/docs/mkdocs/docs/features/binary_formats/cbor.md b/docs/mkdocs/docs/features/binary_formats/cbor.md index e4c257e27..8e6acf0fb 100644 --- a/docs/mkdocs/docs/features/binary_formats/cbor.md +++ b/docs/mkdocs/docs/features/binary_formats/cbor.md @@ -188,7 +188,7 @@ The library maps CBOR types to JSON value types as follows: !!! warning "Tagged items" - Tagged items (0xC0..0xDB) will throw a parse error by default. They can be ignored by passing `cbor_tag_handler_t::ignore` to function `from_cbor`, in which case the tag is skipped and the enclosed data item is parsed on its own. They can be stored by passing `cbor_tag_handler_t::store` to function `from_cbor`. Note that no tag is ever interpreted: for instance, a text string tagged with tag 0 (date/time) stays a string. + Tagged items (0xC0..0xDB) will throw a parse error by default. They can be ignored by passing `cbor_tag_handler_t::ignore` to function `from_cbor`, in which case the tag is skipped and the enclosed data item is parsed on its own. Passing `cbor_tag_handler_t::store` to function `from_cbor` stores tagged byte strings (for bytes 0xd8..0xdb) as binary values with the tag as subtype; other tagged values are read as if the tag were ignored. If several tags precede a byte string, only the innermost one is stored. Note that no tag is ever interpreted: for instance, a text string tagged with tag 0 (date/time) stays a string. ??? example diff --git a/include/nlohmann/detail/input/binary_reader.hpp b/include/nlohmann/detail/input/binary_reader.hpp index 4132c03ca..b6c219fc1 100644 --- a/include/nlohmann/detail/input/binary_reader.hpp +++ b/include/nlohmann/detail/input/binary_reader.hpp @@ -44,7 +44,7 @@ enum class cbor_tag_handler_t { error, ///< throw a parse_error exception in case of a tag ignore, ///< ignore tags - store ///< store tags as binary type + store ///< store tagged byte strings (for bytes 0xd8..0xdb) as binary values with the tag as subtype; other tagged values are read as if the tag were ignored }; /*! @@ -592,14 +592,18 @@ class binary_reader input (true) or whether the last read character should be considered instead (false) @param[in] tag_handler how CBOR tags should be treated + @param[out] tag_pending whether a tag was parsed and its value follows + @param[out] item_read whether the tagged value's initial byte is already in current @return whether a valid CBOR value was passed to the SAX parser */ bool parse_cbor_value(const bool get_char, const cbor_tag_handler_t tag_handler, - bool& tag_pending) + bool& tag_pending, + bool& item_read) { tag_pending = false; + item_read = false; switch (get_char ? get() : current) { @@ -1021,7 +1025,17 @@ class binary_reader } } get(); - return get_cbor_binary(b) && sax->binary(b); + // a byte string (the heads accepted by get_cbor_binary) keeps the tag as subtype + if ((current >= 0x40 && current <= 0x5B) || current == 0x5F) + { + return get_cbor_binary(b) && sax->binary(b); + } + + // not a byte string: the tagged value, whose first byte + // was just read, is read by the caller like for ignore + tag_pending = true; + item_read = true; + return true; } default: // LCOV_EXCL_LINE @@ -1503,13 +1517,14 @@ class binary_reader // a tag is not a value of its own: read on until the tagged value bool tag_pending = false; + bool item_read = false; do { - if (JSON_HEDLEY_UNLIKELY(!parse_cbor_value(fetch, tag_handler, tag_pending))) + if (JSON_HEDLEY_UNLIKELY(!parse_cbor_value(fetch, tag_handler, tag_pending, item_read))) { return false; } - fetch = true; + fetch = !item_read; } while (tag_pending); diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index bf1aee332..81ddfb1c4 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -12741,7 +12741,7 @@ enum class cbor_tag_handler_t { error, ///< throw a parse_error exception in case of a tag ignore, ///< ignore tags - store ///< store tags as binary type + store ///< store tagged byte strings (for bytes 0xd8..0xdb) as binary values with the tag as subtype; other tagged values are read as if the tag were ignored }; /*! @@ -13289,14 +13289,18 @@ class binary_reader input (true) or whether the last read character should be considered instead (false) @param[in] tag_handler how CBOR tags should be treated + @param[out] tag_pending whether a tag was parsed and its value follows + @param[out] item_read whether the tagged value's initial byte is already in current @return whether a valid CBOR value was passed to the SAX parser */ bool parse_cbor_value(const bool get_char, const cbor_tag_handler_t tag_handler, - bool& tag_pending) + bool& tag_pending, + bool& item_read) { tag_pending = false; + item_read = false; switch (get_char ? get() : current) { @@ -13718,7 +13722,17 @@ class binary_reader } } get(); - return get_cbor_binary(b) && sax->binary(b); + // a byte string (the heads accepted by get_cbor_binary) keeps the tag as subtype + if ((current >= 0x40 && current <= 0x5B) || current == 0x5F) + { + return get_cbor_binary(b) && sax->binary(b); + } + + // not a byte string: the tagged value, whose first byte + // was just read, is read by the caller like for ignore + tag_pending = true; + item_read = true; + return true; } default: // LCOV_EXCL_LINE @@ -14200,13 +14214,14 @@ class binary_reader // a tag is not a value of its own: read on until the tagged value bool tag_pending = false; + bool item_read = false; do { - if (JSON_HEDLEY_UNLIKELY(!parse_cbor_value(fetch, tag_handler, tag_pending))) + if (JSON_HEDLEY_UNLIKELY(!parse_cbor_value(fetch, tag_handler, tag_pending, item_read))) { return false; } - fetch = true; + fetch = !item_read; } while (tag_pending); diff --git a/tests/src/unit-cbor.cpp b/tests/src/unit-cbor.cpp index 77f9a10e6..6bd792f8a 100644 --- a/tests/src/unit-cbor.cpp +++ b/tests/src/unit-cbor.cpp @@ -2124,6 +2124,20 @@ TEST_CASE("CBOR nesting does not consume the call stack") CHECK(json::from_cbor(input, true, false, json::cbor_tag_handler_t::ignore).is_discarded()); } + SECTION("stored tags") + { + // a tag over something other than a byte string is read like for + // ignore, so a chain of them must not recurse either (#5316) + std::vector input; + for (std::size_t i = 0; i < 500000; ++i) + { + input.push_back(0xD8); + input.push_back(0x18); + } + input.push_back(0x01); + CHECK(json::from_cbor(input, true, true, json::cbor_tag_handler_t::store) == 1); + } + SECTION("a well-formed deep value is read through the SAX interface") { std::vector input(200000, 0x9F); @@ -3054,6 +3068,77 @@ TEST_CASE("Tagged values") CHECK_THROWS_AS(_ = json::from_cbor(v_tagged, true, true, json::cbor_tag_handler_t::error), json::parse_error); CHECK_THROWS_AS(_ = json::from_cbor(v_tagged, true, true, json::cbor_tag_handler_t::ignore), json::parse_error); } + + SECTION("issue #5316 - cbor_tag_handler_t::store on non-binary tagged items") + { + // 55799({"a": 1}) -- CBOR self-describe magic followed by a map + const std::vector v_map{0xD9, 0xD9, 0xF7, 0xA1, 0x61, 0x61, 0x01}; + CHECK(json::from_cbor(v_map, true, true, json::cbor_tag_handler_t::ignore) == json({{"a", 1}})); + CHECK(json::from_cbor(v_map, true, true, json::cbor_tag_handler_t::store) == json({{"a", 1}})); + + // Tag 24 over unsigned integer 5 + const std::vector v_int{0xD8, 0x18, 0x05}; + CHECK(json::from_cbor(v_int, true, true, json::cbor_tag_handler_t::ignore) == 5); + CHECK(json::from_cbor(v_int, true, true, json::cbor_tag_handler_t::store) == 5); + + // Tag 24 over text string "foo" + const std::vector v_str{0xD8, 0x18, 0x63, 'f', 'o', 'o'}; + CHECK(json::from_cbor(v_str, true, true, json::cbor_tag_handler_t::ignore) == "foo"); + CHECK(json::from_cbor(v_str, true, true, json::cbor_tag_handler_t::store) == "foo"); + + // Tag 24 over array [1, 2] + const std::vector v_arr{0xD8, 0x18, 0x82, 0x01, 0x02}; + CHECK(json::from_cbor(v_arr, true, true, json::cbor_tag_handler_t::ignore) == json({1, 2})); + CHECK(json::from_cbor(v_arr, true, true, json::cbor_tag_handler_t::store) == json({1, 2})); + + // Tag 24 over boolean true + const std::vector v_bool{0xD8, 0x18, 0xF5}; + CHECK(json::from_cbor(v_bool, true, true, json::cbor_tag_handler_t::ignore) == true); + CHECK(json::from_cbor(v_bool, true, true, json::cbor_tag_handler_t::store) == true); + + // Tag 24 over null + const std::vector v_null{0xD8, 0x18, 0xF6}; + CHECK(json::from_cbor(v_null, true, true, json::cbor_tag_handler_t::ignore) == nullptr); + CHECK(json::from_cbor(v_null, true, true, json::cbor_tag_handler_t::store) == nullptr); + + // Nested tags: tag 55799 over tag 24 over integer 42 + const std::vector v_nested{0xD9, 0xD9, 0xF7, 0xD8, 0x18, 0x18, 0x2A}; + CHECK(json::from_cbor(v_nested, true, true, json::cbor_tag_handler_t::ignore) == 42); + CHECK(json::from_cbor(v_nested, true, true, json::cbor_tag_handler_t::store) == 42); + + // Tag 24 over byte string continues to store subtype as before + const std::vector v_bin{0xD8, 0x18, 0x42, 0xCA, 0xFE}; + auto j_bin_store = json::from_cbor(v_bin, true, true, json::cbor_tag_handler_t::store); + CHECK(j_bin_store.is_binary()); + CHECK(j_bin_store.get_binary().has_subtype()); + CHECK(j_bin_store.get_binary().subtype() == 24); + CHECK(j_bin_store.get_binary() == json::binary({0xCA, 0xFE}, 24).get_binary()); + + // Tagged values inside a container under store: [24(1), 25(h'0001')] + const std::vector v_container{0x82, 0xD8, 0x18, 0x01, 0xD8, 0x19, 0x42, 0x00, 0x01}; + auto j_container_store = json::from_cbor(v_container, true, true, json::cbor_tag_handler_t::store); + CHECK(j_container_store.is_array()); + CHECK(j_container_store.size() == 2); + CHECK(j_container_store[0] == 1); + CHECK(j_container_store[1].is_binary()); + CHECK(j_container_store[1].get_binary().has_subtype()); + CHECK(j_container_store[1].get_binary().subtype() == 25); + CHECK(j_container_store[1].get_binary() == json::binary({0x00, 0x01}, 25).get_binary()); + + // Tagged values as object values under store: {"a": 55799(1), "b": 24(h'01')} + const std::vector v_object{0xA2, 0x61, 'a', 0xD9, 0xD9, 0xF7, 0x01, 0x61, 'b', 0xD8, 0x18, 0x41, 0x01}; + CHECK(json::from_cbor(v_object, true, true, json::cbor_tag_handler_t::store) == json({{"a", 1}, {"b", json::binary({0x01}, 24)}})); + + // two tags in a row before a byte string: the inner tag is stored + // (this uses item_read and then the byte-string path) + const std::vector v_nested_byte_string{0xD8, 0x18, 0xD8, 0x19, 0x42, 0x00, 0x01}; + CHECK(json::from_cbor(v_nested_byte_string, true, true, json::cbor_tag_handler_t::store) == json::binary({0x00, 0x01}, 25)); + + // errors after a stored tag are now the same as with ignore + json _; + CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector {0xD8, 0x18}, true, true, json::cbor_tag_handler_t::store), "[json.exception.parse_error.110] parse error at byte 3: syntax error while parsing CBOR value: unexpected end of input", json::parse_error&); + CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector {0xD8, 0x18, 0x1C}, true, true, json::cbor_tag_handler_t::store), "[json.exception.parse_error.112] parse error at byte 3: syntax error while parsing CBOR value: invalid byte: 0x1C", json::parse_error&); + } } SECTION("negative integer overflow") From f682cd2ef1677e1218d714abb177fa2a93eadccf Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Sun, 27 Sep 2026 15:58:59 +0200 Subject: [PATCH 62/64] Skip the #5515 MessagePack size tests on 32-bit platforms (#5590) The tests fake a container size of UINT32_MAX + 1, which does not fit into a 32-bit std::size_t: MSVC rejects the truncation (C4305/C4309 with /WX), and clang-cl wraps the size to 0 so nothing throws. Guard them with SIZE_MAX > UINT32_MAX like the tests from #5584. Signed-off-by: Niels Lohmann --- tests/src/unit-msgpack.cpp | 3 +++ 1 file changed, 3 insertions(+) diff --git a/tests/src/unit-msgpack.cpp b/tests/src/unit-msgpack.cpp index cfb047495..e7616b27f 100644 --- a/tests/src/unit-msgpack.cpp +++ b/tests/src/unit-msgpack.cpp @@ -2164,6 +2164,8 @@ TEST_CASE("MessagePack with std::byte") } #endif +// the fake sizes below do not fit into a 32-bit std::size_t +#if SIZE_MAX > UINT32_MAX template> struct huge_array : std::vector { @@ -2330,6 +2332,7 @@ TEST_CASE("MessagePack Size above uint32 for binary") "[json.exception.out_of_range.412] MessagePack length 4294967296 exceeds maximum of 4294967295", json::out_of_range&); } +#endif namespace { From 1e101ecac1b8dbf2bc1280251d1addc606d4ff8a Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Sun, 27 Sep 2026 16:56:21 +0200 Subject: [PATCH 63/64] Add BON8 support (#2998) * Add BON8 support Add to_bon8/from_bon8 and input_format_t::bon8 for BON8, a binary format that uses the byte values that cannot begin a UTF-8 character as type markers, so strings need no length prefix. It is the most compact of the supported binary formats on the benchmark files. The reader is non-recursive like the other binary readers. A string ends at the first byte that cannot continue it, so the reader hands the one or two bytes it reads past a string back to the value that follows. The writer produces the canonical representation of the specification, except for NFC normalization; its output is identical to that of the reference implementation (HikoGUI) on all files of the test data. The round-trip tests need the .bon8 files of json_test_data 3.2.0. Signed-off-by: Niels Lohmann * Address review comments - Reuse detail::validate_one_utf8 to check strings in to_bon8; the error now names the first byte of the invalid sequence. - Document that to_bon8 leaves bytes in the output adapter on an exception, and that string_open is only an output of write_bon8_marker. - Explain why the pushback buffer of the BON8 reader cannot overflow. Signed-off-by: Niels Lohmann * Select the BON8 float prefix by type get_bon8_float_prefix only depends on the type of its argument, so make the type a template parameter instead of passing an unused value. Signed-off-by: Niels Lohmann * Rename a test variable that Flawfinder mistakes for read() Signed-off-by: Niels Lohmann * Fix the BON8 CI failures - compare the float in write_bon8_float with number_float_t constants, so GCC does not warn about a float-to-double conversion - mark check_bon8_utf8's context as used when exceptions are disabled - choose the compact float prefix in a helper rather than with nested conditional operators (clang-tidy) - use auto for the cast in the BON8 integer reader (clang-tidy) - write the int32 minimum test values as long long literals (MSVC C4146) Signed-off-by: Niels Lohmann * Amalgamate Signed-off-by: Niels Lohmann * Read BON8 strings in bulk from contiguous input - copy the valid UTF-8 of a string in one step when the input is contiguous (twitter.json is read in 1.68 instead of 2.52 ms, jeopardy.json in 196 instead of 297 ms, close to CBOR and MessagePack) - share the new valid_utf8_prefix() with the writer's UTF-8 check, which now skips ASCII 8 bytes at a time - let the fuzzer check that contiguous and stream input give the same value or error, and test both paths in the unit tests - clarify that a second 0xFF after a string is an empty string Signed-off-by: Niels Lohmann * Link the BON8 functions from the other binary format pages Signed-off-by: Niels Lohmann * Name the bulk scan flag after the input, not BON8 Signed-off-by: Niels Lohmann * Read BSON keys in bulk from contiguous input BSON keys (and array indices) are C-style strings, which were read byte by byte. For contiguous input they are now read up to their \x00-byte in one step, using the same bulk_scan flag as BON8 strings: twitter.json is read in 1.46 instead of 2.01 ms, citm_catalog.json in 2.93 instead of 3.33 ms, jeopardy.json in 182 instead of 207 ms. canada.json, whose keys are almost all one-digit array indices, takes 2 % longer. Signed-off-by: Niels Lohmann * Fix the BON8 CI failures of the bulk-read tests - skip the contiguous-versus-stream tests of BON8 strings and BSON keys when exceptions are disabled: they catch the parse errors of invalid input, and without exceptions the library aborts instead - use static_cast for the int64 test value (google-readability-casting) Signed-off-by: Niels Lohmann * Move the explicit basic_json instantiation into its own test file Linking test-regression3_cpp20 with clang and MinGW failed with "relocation truncated to fit: IMAGE_REL_AMD64_REL32 against `.rdata'", as test-regression2 did before #5511. The explicit instantiation of basic_json<> for #4825 compiles every member function, including the BON8 reader and writer, into that object, and it was already close to the limit (2,226,104 bytes on develop, 2,234,960 with BON8; clang -O1, C++20). Give the instantiation a file of its own: unit-regression3 is now 1,594,736 bytes and unit-explicit_instantiation 1,095,064. The new file mentions JSON_HAS_CPP_17 and JSON_HAS_CPP_20 so it keeps being built for the C++17 standard the regression was about. Signed-off-by: Niels Lohmann * Convert the bytes of the BON8 test strings explicitly The str() helper constructed a std::string from a byte range, which converts each unsigned char implicitly; -fsanitize=integer reports that for bytes of 0x80 and above (ci_test_clang_sanitizer). Signed-off-by: Niels Lohmann --------- Signed-off-by: Niels Lohmann --- .github/labeler.yml | 8 +- Makefile | 9 + README.md | 16 +- cmake/ci.cmake | 2 +- cmake/download_test_data.cmake | 2 +- docs/docset/docSet.sql | 3 + .../mkdocs/docs/api/basic_json/from_bjdata.md | 1 + docs/mkdocs/docs/api/basic_json/from_bon8.md | 108 ++ docs/mkdocs/docs/api/basic_json/from_bson.md | 1 + docs/mkdocs/docs/api/basic_json/from_cbor.md | 1 + .../docs/api/basic_json/from_msgpack.md | 1 + .../mkdocs/docs/api/basic_json/from_ubjson.md | 1 + docs/mkdocs/docs/api/basic_json/index.md | 2 + .../docs/api/basic_json/input_format_t.md | 6 +- .../mkdocs/docs/api/basic_json/parse_error.md | 2 +- docs/mkdocs/docs/api/basic_json/sax_parse.md | 3 +- docs/mkdocs/docs/api/basic_json/to_bjdata.md | 1 + docs/mkdocs/docs/api/basic_json/to_bon8.md | 76 ++ docs/mkdocs/docs/api/basic_json/to_bson.md | 1 + docs/mkdocs/docs/api/basic_json/to_cbor.md | 1 + docs/mkdocs/docs/api/basic_json/to_msgpack.md | 1 + docs/mkdocs/docs/api/basic_json/to_ubjson.md | 1 + .../api/macros/json_strict_nul_handling.md | 6 +- docs/mkdocs/docs/examples/from_bon8.cpp | 21 + docs/mkdocs/docs/examples/from_bon8.output | 5 + docs/mkdocs/docs/examples/to_bon8.cpp | 22 + docs/mkdocs/docs/examples/to_bon8.output | 1 + .../docs/features/binary_formats/bon8.md | 159 +++ .../docs/features/binary_formats/index.md | 4 + docs/mkdocs/docs/features/binary_values.md | 35 + docs/mkdocs/docs/features/index.md | 4 +- docs/mkdocs/docs/features/serialization.md | 2 +- .../features/types/template_parameters.md | 2 +- docs/mkdocs/includes/glossary.md | 1 + docs/mkdocs/mkdocs.yml | 5 +- .../nlohmann/detail/input/binary_reader.hpp | 615 +++++++++- .../nlohmann/detail/input/input_adapters.hpp | 2 +- include/nlohmann/detail/input/string_scan.hpp | 37 + .../nlohmann/detail/output/binary_writer.hpp | 364 +++++- include/nlohmann/json.hpp | 61 + single_include/nlohmann/json.hpp | 1081 ++++++++++++++++- tests/CMakeLists.txt | 2 +- tests/Makefile | 5 +- tests/benchmarks/src/benchmarks.cpp | 18 +- tests/fuzzing.md | 6 +- tests/src/fuzzer-parse_bon8.cpp | 103 ++ tests/src/unit-alt-string.cpp | 1 + tests/src/unit-binary_formats.cpp | 15 + tests/src/unit-binary_writer_sinks.cpp | 18 + tests/src/unit-bon8.cpp | 1050 ++++++++++++++++ tests/src/unit-bson.cpp | 39 + tests/src/unit-class_parser.cpp | 6 +- tests/src/unit-custom-binary-type.cpp | 2 + tests/src/unit-custom-object-type.cpp | 1 + tests/src/unit-explicit_instantiation.cpp | 37 + tests/src/unit-ordered_json2.cpp | 16 + tests/src/unit-regression2.cpp | 1 + tests/src/unit-regression3.cpp | 9 +- 58 files changed, 3950 insertions(+), 53 deletions(-) create mode 100644 docs/mkdocs/docs/api/basic_json/from_bon8.md create mode 100644 docs/mkdocs/docs/api/basic_json/to_bon8.md create mode 100644 docs/mkdocs/docs/examples/from_bon8.cpp create mode 100644 docs/mkdocs/docs/examples/from_bon8.output create mode 100644 docs/mkdocs/docs/examples/to_bon8.cpp create mode 100644 docs/mkdocs/docs/examples/to_bon8.output create mode 100644 docs/mkdocs/docs/features/binary_formats/bon8.md create mode 100644 tests/src/fuzzer-parse_bon8.cpp create mode 100644 tests/src/unit-bon8.cpp create mode 100644 tests/src/unit-explicit_instantiation.cpp diff --git a/.github/labeler.yml b/.github/labeler.yml index b4c176960..5708f84f5 100644 --- a/.github/labeler.yml +++ b/.github/labeler.yml @@ -37,13 +37,13 @@ labels: files: - "include/nlohmann/detail/input/binary_reader\\.hpp" - "include/nlohmann/detail/output/binary_writer\\.hpp" - - "tests/src/unit-(bson|cbor|msgpack|ubjson|bjdata|binary_formats)" - - "tests/src/fuzzer-parse_(bson|cbor|msgpack|ubjson|bjdata)" + - "tests/src/unit-(bson|cbor|msgpack|ubjson|bjdata|bon8|binary_formats)" + - "tests/src/fuzzer-parse_(bson|cbor|msgpack|ubjson|bjdata|bon8)" - "docs/mkdocs/docs/features/binary_formats/" - - "docs/mkdocs/docs/(api/basic_json|examples)/(to|from)_(bson|cbor|msgpack|ubjson|bjdata)" + - "docs/mkdocs/docs/(api/basic_json|examples)/(to|from)_(bson|cbor|msgpack|ubjson|bjdata|bon8)" - label: "aspect: binary formats" - title: "(?i)(bson|cbor|msgpack|messagepack|ubjson|bjdata|binary format)" + title: "(?i)(bson|cbor|msgpack|messagepack|ubjson|bjdata|bon8|binary format)" - label: "python" files: diff --git a/Makefile b/Makefile index 871ea7995..196f87243 100644 --- a/Makefile +++ b/Makefile @@ -36,6 +36,7 @@ all: @echo "clean - remove built files" @echo "doctest - compile example files and check their output" @echo "fuzz_testing - prepare fuzz testing of the JSON parser" + @echo "fuzz_testing_bon8 - prepare fuzz testing of the BON8 parser" @echo "fuzz_testing_bson - prepare fuzz testing of the BSON parser" @echo "fuzz_testing_cbor - prepare fuzz testing of the CBOR parser" @echo "fuzz_testing_msgpack - prepare fuzz testing of the MessagePack parser" @@ -71,6 +72,14 @@ fuzz_testing: find tests/data/json_tests -size -5k -name *json | xargs -I{} cp "{}" fuzz-testing/testcases @echo "Execute: afl-fuzz -i fuzz-testing/testcases -o fuzz-testing/out fuzz-testing/fuzzer" +fuzz_testing_bon8: + rm -fr fuzz-testing + mkdir -p fuzz-testing fuzz-testing/testcases fuzz-testing/out + $(MAKE) parse_bon8_fuzzer -C tests CXX=afl-clang++ + mv tests/parse_bon8_fuzzer fuzz-testing/fuzzer + find tests/data -size -5k -name *.bon8 | xargs -I{} cp "{}" fuzz-testing/testcases + @echo "Execute: afl-fuzz -i fuzz-testing/testcases -o fuzz-testing/out fuzz-testing/fuzzer" + fuzz_testing_bson: rm -fr fuzz-testing mkdir -p fuzz-testing fuzz-testing/testcases fuzz-testing/out diff --git a/README.md b/README.md index 412d05fd4..cc3546686 100644 --- a/README.md +++ b/README.md @@ -40,7 +40,7 @@ - [Implicit conversions](#implicit-conversions) - [Conversions to/from arbitrary types](#arbitrary-types-conversions) - [Specializing enum conversion](#specializing-enum-conversion) - - [Binary formats (BSON, CBOR, MessagePack, UBJSON, and BJData)](#binary-formats-bson-cbor-messagepack-ubjson-and-bjdata) + - [Binary formats (BSON, CBOR, MessagePack, UBJSON, BJData, and BON8)](#binary-formats-bson-cbor-messagepack-ubjson-bjdata-and-bon8) - [Customers](#customers) - [Ecosystem](#ecosystem) - [Supported compilers](#supported-compilers) @@ -128,7 +128,7 @@ There is also a [**docset**](https://github.com/Kapeli/Dash-User-Contributions/t - **JSON Pointer functions**: [flatten](https://json.nlohmann.me/api/basic_json/flatten), [unflatten](https://json.nlohmann.me/api/basic_json/unflatten) - **JSON Patch functions**: [patch](https://json.nlohmann.me/api/basic_json/patch), [patch_inplace](https://json.nlohmann.me/api/basic_json/patch_inplace), [diff](https://json.nlohmann.me/api/basic_json/diff), [merge_patch](https://json.nlohmann.me/api/basic_json/merge_patch) - **Static functions**: [meta](https://json.nlohmann.me/api/basic_json/meta), [get_allocator](https://json.nlohmann.me/api/basic_json/get_allocator) -- **Binary formats**: [from_bjdata](https://json.nlohmann.me/api/basic_json/from_bjdata), [from_bson](https://json.nlohmann.me/api/basic_json/from_bson), [from_cbor](https://json.nlohmann.me/api/basic_json/from_cbor), [from_msgpack](https://json.nlohmann.me/api/basic_json/from_msgpack), [from_ubjson](https://json.nlohmann.me/api/basic_json/from_ubjson), [to_bjdata](https://json.nlohmann.me/api/basic_json/to_bjdata), [to_bson](https://json.nlohmann.me/api/basic_json/to_bson), [to_cbor](https://json.nlohmann.me/api/basic_json/to_cbor), [to_msgpack](https://json.nlohmann.me/api/basic_json/to_msgpack), [to_ubjson](https://json.nlohmann.me/api/basic_json/to_ubjson) +- **Binary formats**: [from_bjdata](https://json.nlohmann.me/api/basic_json/from_bjdata), [from_bon8](https://json.nlohmann.me/api/basic_json/from_bon8), [from_bson](https://json.nlohmann.me/api/basic_json/from_bson), [from_cbor](https://json.nlohmann.me/api/basic_json/from_cbor), [from_msgpack](https://json.nlohmann.me/api/basic_json/from_msgpack), [from_ubjson](https://json.nlohmann.me/api/basic_json/from_ubjson), [to_bjdata](https://json.nlohmann.me/api/basic_json/to_bjdata), [to_bon8](https://json.nlohmann.me/api/basic_json/to_bon8), [to_bson](https://json.nlohmann.me/api/basic_json/to_bson), [to_cbor](https://json.nlohmann.me/api/basic_json/to_cbor), [to_msgpack](https://json.nlohmann.me/api/basic_json/to_msgpack), [to_ubjson](https://json.nlohmann.me/api/basic_json/to_ubjson) - **Non-member functions**: [operator<<](https://json.nlohmann.me/api/operator_ltlt/), [operator>>](https://json.nlohmann.me/api/operator_gtgt/), [to_string](https://json.nlohmann.me/api/basic_json/to_string) - **Literals**: [operator""_json](https://json.nlohmann.me/api/operator_literal_json) - **Helper classes**: [std::hash<basic_json>](https://json.nlohmann.me/api/basic_json/std_hash), [std::swap<basic_json>](https://json.nlohmann.me/api/basic_json/std_swap) @@ -1110,9 +1110,9 @@ Other Important points: - When using `get()`, undefined JSON values will default to the first pair specified in your map. Select this default pair carefully. If you desire an exception in this circumstance use `NLOHMANN_JSON_SERIALIZE_ENUM_STRICT()` which behaves identically except for throwing an exception on unrecognized values. - If an enum or JSON value is specified more than once in your map, the first matching occurrence from the top of the map will be returned when converting to or from JSON. -### Binary formats (BSON, CBOR, MessagePack, UBJSON, and BJData) +### Binary formats (BSON, CBOR, MessagePack, UBJSON, BJData, and BON8) -Though JSON is a ubiquitous data format, it is not a very compact format suitable for data exchange, for instance over a network. Hence, the library supports [BSON](https://bsonspec.org) (Binary JSON), [CBOR](https://cbor.io) (Concise Binary Object Representation), [MessagePack](https://msgpack.org), [UBJSON](https://ubjson.org) (Universal Binary JSON Specification) and [BJData](https://neurojson.org/bjdata) (Binary JData) to efficiently encode JSON values to byte vectors and to decode such vectors. +Though JSON is a ubiquitous data format, it is not a very compact format suitable for data exchange, for instance over a network. Hence, the library supports [BSON](https://bsonspec.org) (Binary JSON), [CBOR](https://cbor.io) (Concise Binary Object Representation), [MessagePack](https://msgpack.org), [UBJSON](https://ubjson.org) (Universal Binary JSON Specification), [BJData](https://neurojson.org/bjdata) (Binary JData), and [BON8](https://github.com/hikoworks/hikogui/blob/main/docs/BON8.md) (Binary Object Notation 8) to efficiently encode JSON values to byte vectors and to decode such vectors. ```cpp // create a JSON value @@ -1149,6 +1149,14 @@ std::vector v_ubjson = json::to_ubjson(j); // roundtrip json j_from_ubjson = json::from_ubjson(v_ubjson); + +// serialize to BON8 +std::vector v_bon8 = json::to_bon8(j); + +// 0x88, 0x63, 0x6F, 0x6D, 0x70, 0x61, 0x63, 0x74, 0xF9, 0x73, 0x63, 0x68, 0x65, 0x6D, 0x61, 0x90 + +// roundtrip +json j_from_bon8 = json::from_bon8(v_bon8); ``` The library also supports binary types from BSON, CBOR (byte strings), and MessagePack (bin, ext, fixext). They are stored by default as `std::vector` to be processed outside the library. diff --git a/cmake/ci.cmake b/cmake/ci.cmake index 573a8e0e0..f7695c36e 100644 --- a/cmake/ci.cmake +++ b/cmake/ci.cmake @@ -542,7 +542,7 @@ add_custom_target(ci_infer add_custom_target(ci_offline_testdata COMMAND mkdir -p ${PROJECT_BINARY_DIR}/build_offline_testdata/test_data - COMMAND cd ${PROJECT_BINARY_DIR}/build_offline_testdata/test_data && ${GIT_TOOL} clone -c advice.detachedHead=false --branch v3.1.0 https://github.com/nlohmann/json_test_data.git --quiet --depth 1 + COMMAND cd ${PROJECT_BINARY_DIR}/build_offline_testdata/test_data && ${GIT_TOOL} clone -c advice.detachedHead=false --branch v3.2.0 https://github.com/nlohmann/json_test_data.git --quiet --depth 1 COMMAND ${CMAKE_COMMAND} -DCMAKE_BUILD_TYPE=Debug -GNinja -DJSON_BuildTests=ON -DJSON_FastTests=ON -DJSON_TestDataDirectory=${PROJECT_BINARY_DIR}/build_offline_testdata/test_data/json_test_data diff --git a/cmake/download_test_data.cmake b/cmake/download_test_data.cmake index 3b6f9a395..984c445be 100644 --- a/cmake/download_test_data.cmake +++ b/cmake/download_test_data.cmake @@ -1,5 +1,5 @@ set(JSON_TEST_DATA_URL https://github.com/nlohmann/json_test_data) -set(JSON_TEST_DATA_VERSION 3.1.0) +set(JSON_TEST_DATA_VERSION 3.2.0) include(ExternalProject) diff --git a/docs/docset/docSet.sql b/docs/docset/docSet.sql index 38b203d38..3a97e405f 100644 --- a/docs/docset/docSet.sql +++ b/docs/docset/docSet.sql @@ -48,6 +48,7 @@ INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::from_bjdata', 'Fu INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::from_bson', 'Function', 'api/basic_json/from_bson/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::from_cbor', 'Function', 'api/basic_json/from_cbor/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::from_msgpack', 'Function', 'api/basic_json/from_msgpack/index.html'); +INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::from_bon8', 'Function', 'api/basic_json/from_bon8/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::from_ubjson', 'Function', 'api/basic_json/from_ubjson/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::front', 'Method', 'api/basic_json/front/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::get', 'Method', 'api/basic_json/get/index.html'); @@ -121,6 +122,7 @@ INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::to_bjdata', 'Func INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::to_bson', 'Function', 'api/basic_json/to_bson/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::to_cbor', 'Function', 'api/basic_json/to_cbor/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::to_msgpack', 'Function', 'api/basic_json/to_msgpack/index.html'); +INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::to_bon8', 'Function', 'api/basic_json/to_bon8/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::to_string', 'Method', 'api/basic_json/to_string/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::to_ubjson', 'Function', 'api/basic_json/to_ubjson/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::value', 'Method', 'api/basic_json/value/index.html'); @@ -171,6 +173,7 @@ INSERT INTO searchIndex(name, type, path) VALUES ('Binary Formats: BJData', 'Gui INSERT INTO searchIndex(name, type, path) VALUES ('Binary Formats: BSON', 'Guide', 'features/binary_formats/bson/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('Binary Formats: CBOR', 'Guide', 'features/binary_formats/cbor/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('Binary Formats: MessagePack', 'Guide', 'features/binary_formats/messagepack/index.html'); +INSERT INTO searchIndex(name, type, path) VALUES ('Binary Formats: BON8', 'Guide', 'features/binary_formats/bon8/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('Binary Formats: UBJSON', 'Guide', 'features/binary_formats/ubjson/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('Binary Values', 'Guide', 'features/binary_values/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('Comments', 'Guide', 'features/comments/index.html'); diff --git a/docs/mkdocs/docs/api/basic_json/from_bjdata.md b/docs/mkdocs/docs/api/basic_json/from_bjdata.md index ac60bb58b..9df21ab30 100644 --- a/docs/mkdocs/docs/api/basic_json/from_bjdata.md +++ b/docs/mkdocs/docs/api/basic_json/from_bjdata.md @@ -104,6 +104,7 @@ Linear in the size of the input. - [from_msgpack](from_msgpack.md) create a JSON value from an input in MessagePack format - [from_bson](from_bson.md) create a JSON value from an input in BSON format - [from_ubjson](from_ubjson.md) create a JSON value from an input in UBJSON format +- [from_bon8](from_bon8.md) create a JSON value from an input in BON8 format ## Version history diff --git a/docs/mkdocs/docs/api/basic_json/from_bon8.md b/docs/mkdocs/docs/api/basic_json/from_bon8.md new file mode 100644 index 000000000..2f9d1e749 --- /dev/null +++ b/docs/mkdocs/docs/api/basic_json/from_bon8.md @@ -0,0 +1,108 @@ +# nlohmann::basic_json::from_bon8 + +```cpp +// (1) +template +static basic_json from_bon8(InputType&& i, + const bool strict = true, + const bool allow_exceptions = true); +// (2) +template +static basic_json from_bon8(IteratorType first, SentinelType last, + const bool strict = true, + const bool allow_exceptions = true); +``` + +Deserializes a given input to a JSON value using the BON8 (Binary Object Notation 8) serialization format. + +1. Reads from a compatible input. +2. Reads from an iterator range, or an iterator and a sentinel of a different type (C++20 ranges support). + +The exact mapping and its limitations are described on a [dedicated page](../../features/binary_formats/bon8.md). + +## Template parameters + +`InputType` +: A compatible input, for instance: + + - an `std::istream` object + - a `FILE` pointer + - a C-style array of characters + - a pointer to a null-terminated string of single byte characters + - a container `obj` for which `begin(obj)` and `end(obj)` produce a valid pair of iterators + (as found via ADL or member functions, with semantics compatible to `std::begin` and `std::end`) + +`IteratorType` +: a compatible iterator type + +`SentinelType` +: defaults to `IteratorType`; may be a different type comparable to `IteratorType` via `operator!=`, for instance. + + - a custom sentinel type for C++20 ranges + - `std::default_sentinel_t`, when `IteratorType` is `std::counted_iterator` + +## Parameters + +`i` (in) +: an input in BON8 format convertible to an input adapter + +`first` (in) +: iterator to the start of the input + +`last` (in) +: iterator to the end of the input, or a sentinel value that compares equal to the end iterator with `operator!=` + +`strict` (in) +: whether to expect the input to be consumed until EOF (`#!cpp true` by default) + +`allow_exceptions` (in) +: whether to throw exceptions in case of a parse error (optional, `#!cpp true` by default) + +## Return value + +deserialized JSON value; in case of a parse error and `allow_exceptions` set to `#!cpp false`, the return value will be +`value_t::discarded`. The latter can be checked with [`is_discarded`](is_discarded.md). + +## Exception safety + +Strong guarantee: if an exception is thrown, there are no changes in the JSON value. + +## Exceptions + +- Throws [parse_error.110](../../home/exceptions.md#jsonexceptionparse_error110) if the given input ends prematurely or + the end of the file was not reached when `strict` was set to true +- Throws [parse_error.112](../../home/exceptions.md#jsonexceptionparse_error112) if a parse error occurs, for instance + an invalid byte, a string that is not valid UTF-8, or an object key that is not a string + +## Complexity + +Linear in the size of the input. + +## Examples + +??? example + + The example shows the deserialization of a byte vector in BON8 format to a JSON value. + + ```cpp + --8<-- "examples/from_bon8.cpp" + ``` + + Output: + + ```json + --8<-- "examples/from_bon8.output" + ``` + +## See also + +- [to_bon8](to_bon8.md) create a BON8 serialization of a JSON value +- [from_cbor](from_cbor.md) create a JSON value from an input in CBOR format +- [from_msgpack](from_msgpack.md) create a JSON value from an input in MessagePack format +- [from_bson](from_bson.md) create a JSON value from an input in BSON format +- [from_ubjson](from_ubjson.md) create a JSON value from an input in UBJSON format +- [from_bjdata](from_bjdata.md) create a JSON value from an input in BJData format + +## Version history + +- Added in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json/from_bson.md b/docs/mkdocs/docs/api/basic_json/from_bson.md index ce093a0d8..cf022aeb6 100644 --- a/docs/mkdocs/docs/api/basic_json/from_bson.md +++ b/docs/mkdocs/docs/api/basic_json/from_bson.md @@ -104,6 +104,7 @@ Linear in the size of the input. - [from_msgpack](from_msgpack.md) for the related MessagePack format - [from_ubjson](from_ubjson.md) for the related UBJSON format - [from_bjdata](from_bjdata.md) for the related BJData format +- [from_bon8](from_bon8.md) for the related BON8 format ## Version history diff --git a/docs/mkdocs/docs/api/basic_json/from_cbor.md b/docs/mkdocs/docs/api/basic_json/from_cbor.md index fb28edd3b..b72f55280 100644 --- a/docs/mkdocs/docs/api/basic_json/from_cbor.md +++ b/docs/mkdocs/docs/api/basic_json/from_cbor.md @@ -110,6 +110,7 @@ Linear in the size of the input. - [from_bson](from_bson.md) create a JSON value from an input in BSON format - [from_ubjson](from_ubjson.md) create a JSON value from an input in UBJSON format - [from_bjdata](from_bjdata.md) create a JSON value from an input in BJData format +- [from_bon8](from_bon8.md) create a JSON value from an input in BON8 format ## Version history diff --git a/docs/mkdocs/docs/api/basic_json/from_msgpack.md b/docs/mkdocs/docs/api/basic_json/from_msgpack.md index 12834e2c3..2f4b7bb3b 100644 --- a/docs/mkdocs/docs/api/basic_json/from_msgpack.md +++ b/docs/mkdocs/docs/api/basic_json/from_msgpack.md @@ -103,6 +103,7 @@ Linear in the size of the input. - [from_bson](from_bson.md) create a JSON value from an input in BSON format - [from_ubjson](from_ubjson.md) create a JSON value from an input in UBJSON format - [from_bjdata](from_bjdata.md) create a JSON value from an input in BJData format +- [from_bon8](from_bon8.md) create a JSON value from an input in BON8 format ## Version history diff --git a/docs/mkdocs/docs/api/basic_json/from_ubjson.md b/docs/mkdocs/docs/api/basic_json/from_ubjson.md index a0f8944aa..0d060c750 100644 --- a/docs/mkdocs/docs/api/basic_json/from_ubjson.md +++ b/docs/mkdocs/docs/api/basic_json/from_ubjson.md @@ -104,6 +104,7 @@ Linear in the size of the input. - [from_msgpack](from_msgpack.md) create a JSON value from an input in MessagePack format - [from_bson](from_bson.md) create a JSON value from an input in BSON format - [from_bjdata](from_bjdata.md) create a JSON value from an input in BJData format +- [from_bon8](from_bon8.md) create a JSON value from an input in BON8 format ## Version history diff --git a/docs/mkdocs/docs/api/basic_json/index.md b/docs/mkdocs/docs/api/basic_json/index.md index c5556ea3b..866bda67a 100644 --- a/docs/mkdocs/docs/api/basic_json/index.md +++ b/docs/mkdocs/docs/api/basic_json/index.md @@ -290,11 +290,13 @@ Access to the JSON value ### Binary formats - [**from_bjdata**](from_bjdata.md) (_static_) - create a JSON value from an input in BJData format +- [**from_bon8**](from_bon8.md) (_static_) - create a JSON value from an input in BON8 format - [**from_bson**](from_bson.md) (_static_) - create a JSON value from an input in BSON format - [**from_cbor**](from_cbor.md) (_static_) - create a JSON value from an input in CBOR format - [**from_msgpack**](from_msgpack.md) (_static_) - create a JSON value from an input in MessagePack format - [**from_ubjson**](from_ubjson.md) (_static_) - create a JSON value from an input in UBJSON format - [**to_bjdata**](to_bjdata.md) (_static_) - create a BJData serialization of a given JSON value +- [**to_bon8**](to_bon8.md) (_static_) - create a BON8 serialization of a given JSON value - [**to_bson**](to_bson.md) (_static_) - create a BSON serialization of a given JSON value - [**to_cbor**](to_cbor.md) (_static_) - create a CBOR serialization of a given JSON value - [**to_msgpack**](to_msgpack.md) (_static_) - create a MessagePack serialization of a given JSON value diff --git a/docs/mkdocs/docs/api/basic_json/input_format_t.md b/docs/mkdocs/docs/api/basic_json/input_format_t.md index a3baabab8..407b43427 100644 --- a/docs/mkdocs/docs/api/basic_json/input_format_t.md +++ b/docs/mkdocs/docs/api/basic_json/input_format_t.md @@ -7,7 +7,8 @@ enum class input_format_t { msgpack, ubjson, bson, - bjdata + bjdata, + bon8 }; ``` @@ -31,6 +32,9 @@ bson bjdata : BJData (Binary JData) +bon8 +: BON8 (Binary Object Notation 8) + ## Examples ??? example diff --git a/docs/mkdocs/docs/api/basic_json/parse_error.md b/docs/mkdocs/docs/api/basic_json/parse_error.md index 9931d5463..74e16f41a 100644 --- a/docs/mkdocs/docs/api/basic_json/parse_error.md +++ b/docs/mkdocs/docs/api/basic_json/parse_error.md @@ -5,7 +5,7 @@ class parse_error : public exception; ``` The library throws this exception when a parse error occurs. Parse errors can occur during the deserialization of -JSON text, BSON, CBOR, MessagePack, UBJSON, as well as when using JSON Patch. +JSON text, BJData, BON8, BSON, CBOR, MessagePack, UBJSON, as well as when using JSON Patch. Member `byte` holds the byte index of the last read character in the input file (see note below). diff --git a/docs/mkdocs/docs/api/basic_json/sax_parse.md b/docs/mkdocs/docs/api/basic_json/sax_parse.md index bf61ea9eb..d6ce688a7 100644 --- a/docs/mkdocs/docs/api/basic_json/sax_parse.md +++ b/docs/mkdocs/docs/api/basic_json/sax_parse.md @@ -65,7 +65,8 @@ The SAX event lister must follow the interface of [`json_sax`](../json_sax/index : SAX event listener (must not be null) `format` (in) -: the format to parse (JSON, CBOR, MessagePack, or UBJSON) (optional, `input_format_t::json` by default), see +: the format to parse (JSON, BJData, BON8, BSON, CBOR, MessagePack, or UBJSON) (optional, `input_format_t::json` by + default), see [`input_format_t`](input_format_t.md) for more information `strict` (in) diff --git a/docs/mkdocs/docs/api/basic_json/to_bjdata.md b/docs/mkdocs/docs/api/basic_json/to_bjdata.md index 2cf936a51..b066e6852 100644 --- a/docs/mkdocs/docs/api/basic_json/to_bjdata.md +++ b/docs/mkdocs/docs/api/basic_json/to_bjdata.md @@ -84,6 +84,7 @@ Linear in the size of the JSON value `j`. - [to_msgpack](to_msgpack.md) create a MessagePack serialization of a JSON value - [to_bson](to_bson.md) create a BSON serialization of a JSON value - [to_ubjson](to_ubjson.md) create a UBJSON serialization of a JSON value +- [to_bon8](to_bon8.md) create a BON8 serialization of a JSON value ## Version history diff --git a/docs/mkdocs/docs/api/basic_json/to_bon8.md b/docs/mkdocs/docs/api/basic_json/to_bon8.md new file mode 100644 index 000000000..26d552e8a --- /dev/null +++ b/docs/mkdocs/docs/api/basic_json/to_bon8.md @@ -0,0 +1,76 @@ +# nlohmann::basic_json::to_bon8 + +```cpp +// (1) +static std::vector to_bon8(const basic_json& j); + +// (2) +static void to_bon8(const basic_json& j, detail::output_adapter o); +static void to_bon8(const basic_json& j, detail::output_adapter o); +``` + +Serializes a given JSON value `j` to a byte vector using the BON8 (Binary Object Notation 8) serialization format. BON8 +is a compact binary serialization format that stores strings as UTF-8 without a length prefix. + +1. Returns a byte vector containing the BON8 serialization. +2. Writes the BON8 serialization to an output adapter. + +The exact mapping and its limitations are described on a [dedicated page](../../features/binary_formats/bon8.md). + +## Parameters + +`j` (in) +: JSON value to serialize + +`o` (in) +: output adapter to write serialization to + +## Return value + +1. BON8 serialization as a byte vector +2. (none) + +## Exception safety + +Strong guarantee: if an exception is thrown, there are no changes in the JSON value `j`, which is never modified. +With (2), the bytes written before the exception remain in the output adapter. + +## Exceptions + +- Throws [out_of_range.407](../../home/exceptions.md#jsonexceptionout_of_range407) if `j` contains an unsigned integer + above 9223372036854775807, which BON8 cannot represent +- Throws [type_error.316](../../home/exceptions.md#jsonexceptiontype_error316) if `j` contains a string that is not + valid UTF-8 + +## Complexity + +Linear in the size of the JSON value `j`. + +## Examples + +??? example + + The example shows the serialization of a JSON value to a byte vector in BON8 format. + + ```cpp + --8<-- "examples/to_bon8.cpp" + ``` + + Output: + + ```json + --8<-- "examples/to_bon8.output" + ``` + +## See also + +- [from_bon8](from_bon8.md) create a JSON value from an input in BON8 format +- [to_cbor](to_cbor.md) create a CBOR serialization of a JSON value +- [to_msgpack](to_msgpack.md) create a MessagePack serialization of a JSON value +- [to_bson](to_bson.md) create a BSON serialization of a JSON value +- [to_ubjson](to_ubjson.md) create a UBJSON serialization of a JSON value +- [to_bjdata](to_bjdata.md) create a BJData serialization of a JSON value + +## Version history + +- Added in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json/to_bson.md b/docs/mkdocs/docs/api/basic_json/to_bson.md index 4cd45a57d..fb02c51e1 100644 --- a/docs/mkdocs/docs/api/basic_json/to_bson.md +++ b/docs/mkdocs/docs/api/basic_json/to_bson.md @@ -72,6 +72,7 @@ pass before anything is written. - [to_msgpack](to_msgpack.md) create a MessagePack serialization of a JSON value - [to_ubjson](to_ubjson.md) create a UBJSON serialization of a JSON value - [to_bjdata](to_bjdata.md) create a BJData serialization of a JSON value +- [to_bon8](to_bon8.md) create a BON8 serialization of a JSON value ## Version history diff --git a/docs/mkdocs/docs/api/basic_json/to_cbor.md b/docs/mkdocs/docs/api/basic_json/to_cbor.md index 93facfca0..3bbd9c7d3 100644 --- a/docs/mkdocs/docs/api/basic_json/to_cbor.md +++ b/docs/mkdocs/docs/api/basic_json/to_cbor.md @@ -62,6 +62,7 @@ Linear in the size of the JSON value `j`. - [to_bson](to_bson.md) create a BSON serialization of a JSON value - [to_ubjson](to_ubjson.md) create a UBJSON serialization of a JSON value - [to_bjdata](to_bjdata.md) create a BJData serialization of a JSON value +- [to_bon8](to_bon8.md) create a BON8 serialization of a JSON value ## Version history diff --git a/docs/mkdocs/docs/api/basic_json/to_msgpack.md b/docs/mkdocs/docs/api/basic_json/to_msgpack.md index b3bcaab7f..007fb1914 100644 --- a/docs/mkdocs/docs/api/basic_json/to_msgpack.md +++ b/docs/mkdocs/docs/api/basic_json/to_msgpack.md @@ -70,6 +70,7 @@ Linear in the size of the JSON value `j`. - [to_bson](to_bson.md) create a BSON serialization of a JSON value - [to_ubjson](to_ubjson.md) create a UBJSON serialization of a JSON value - [to_bjdata](to_bjdata.md) create a BJData serialization of a JSON value +- [to_bon8](to_bon8.md) create a BON8 serialization of a JSON value ## Version history diff --git a/docs/mkdocs/docs/api/basic_json/to_ubjson.md b/docs/mkdocs/docs/api/basic_json/to_ubjson.md index 694fbbba6..6437ba3e5 100644 --- a/docs/mkdocs/docs/api/basic_json/to_ubjson.md +++ b/docs/mkdocs/docs/api/basic_json/to_ubjson.md @@ -77,6 +77,7 @@ Linear in the size of the JSON value `j`. - [to_msgpack](to_msgpack.md) create a MessagePack serialization of a JSON value - [to_bson](to_bson.md) create a BSON serialization of a JSON value - [to_bjdata](to_bjdata.md) create a BJData serialization of a JSON value +- [to_bon8](to_bon8.md) create a BON8 serialization of a JSON value ## Version history diff --git a/docs/mkdocs/docs/api/macros/json_strict_nul_handling.md b/docs/mkdocs/docs/api/macros/json_strict_nul_handling.md index a50ce2645..f22395d3b 100644 --- a/docs/mkdocs/docs/api/macros/json_strict_nul_handling.md +++ b/docs/mkdocs/docs/api/macros/json_strict_nul_handling.md @@ -11,9 +11,9 @@ The macro only affects the JSON text parser ([`parse`](../basic_json/parse.md), [`sax_parse`](../basic_json/sax_parse.md), and [`operator>>`](../operator_gtgt.md)). There are three cases where a NUL byte is still not rejected: -- The binary formats ([`from_bjdata`](../basic_json/from_bjdata.md), [`from_bson`](../basic_json/from_bson.md), - [`from_cbor`](../basic_json/from_cbor.md), [`from_msgpack`](../basic_json/from_msgpack.md), - [`from_ubjson`](../basic_json/from_ubjson.md)) are never affected: there, `0x00` is ordinary data. +- The binary formats ([`from_bjdata`](../basic_json/from_bjdata.md), [`from_bon8`](../basic_json/from_bon8.md), + [`from_bson`](../basic_json/from_bson.md), [`from_cbor`](../basic_json/from_cbor.md), + [`from_msgpack`](../basic_json/from_msgpack.md), [`from_ubjson`](../basic_json/from_ubjson.md)) are never affected: there, `0x00` is ordinary data. - A bare `const char*` pointer has no length of its own, so its length is still determined with `strlen()`. The first NUL byte therefore still marks the end of the input, and nothing after it is read. - One trailing `'\0'` at the end of a `char` array (e.g., a string literal) is trimmed; see the warning below. diff --git a/docs/mkdocs/docs/examples/from_bon8.cpp b/docs/mkdocs/docs/examples/from_bon8.cpp new file mode 100644 index 000000000..83ab82060 --- /dev/null +++ b/docs/mkdocs/docs/examples/from_bon8.cpp @@ -0,0 +1,21 @@ +#include +#include +#include + +using json = nlohmann::json; + +int main() +{ + // create byte vector + std::vector v = {0x89, 0x63, 0x6f, 0x6d, 0x70, 0x61, 0x63, 0x74, + 0xf9, 0x66, 0x6f, 0x72, 0x6d, 0x61, 0x74, 0xff, + 0x42, 0x4f, 0x4e, 0x38, 0xff, 0x73, 0x63, 0x68, + 0x65, 0x6d, 0x61, 0x90 + }; + + // deserialize it with BON8 + json j = json::from_bon8(v); + + // print the deserialized JSON value + std::cout << std::setw(2) << j << std::endl; +} diff --git a/docs/mkdocs/docs/examples/from_bon8.output b/docs/mkdocs/docs/examples/from_bon8.output new file mode 100644 index 000000000..78c64c2b4 --- /dev/null +++ b/docs/mkdocs/docs/examples/from_bon8.output @@ -0,0 +1,5 @@ +{ + "compact": true, + "format": "BON8", + "schema": 0 +} diff --git a/docs/mkdocs/docs/examples/to_bon8.cpp b/docs/mkdocs/docs/examples/to_bon8.cpp new file mode 100644 index 000000000..a40130903 --- /dev/null +++ b/docs/mkdocs/docs/examples/to_bon8.cpp @@ -0,0 +1,22 @@ +#include +#include +#include + +using json = nlohmann::json; +using namespace nlohmann::literals; + +int main() +{ + // create a JSON value + json j = R"({"compact": true, "format": "BON8", "schema": 0})"_json; + + // serialize it to BON8 + std::vector v = json::to_bon8(j); + + // print the vector content + for (auto& byte : v) + { + std::cout << "0x" << std::hex << std::setw(2) << std::setfill('0') << (int)byte << " "; + } + std::cout << std::endl; +} diff --git a/docs/mkdocs/docs/examples/to_bon8.output b/docs/mkdocs/docs/examples/to_bon8.output new file mode 100644 index 000000000..05481d51f --- /dev/null +++ b/docs/mkdocs/docs/examples/to_bon8.output @@ -0,0 +1 @@ +0x89 0x63 0x6f 0x6d 0x70 0x61 0x63 0x74 0xf9 0x66 0x6f 0x72 0x6d 0x61 0x74 0xff 0x42 0x4f 0x4e 0x38 0xff 0x73 0x63 0x68 0x65 0x6d 0x61 0x90 diff --git a/docs/mkdocs/docs/features/binary_formats/bon8.md b/docs/mkdocs/docs/features/binary_formats/bon8.md new file mode 100644 index 000000000..9b5fa1dda --- /dev/null +++ b/docs/mkdocs/docs/features/binary_formats/bon8.md @@ -0,0 +1,159 @@ +# BON8 + +BON8 (Binary Object Notation 8) is a compact binary serialization format for JSON values. It uses the byte values that +cannot begin a UTF-8 character as type markers, so strings are stored as plain UTF-8 without a length prefix: a string +ends at the first byte that cannot continue it. Integers from -10 to 39, `true`, `false`, `null`, and the floating-point +values -1.0, 0.0, and 1.0 take a single byte, and arrays and objects with up to four elements need no terminator. + +!!! abstract "References" + + - [BON8 specification](https://github.com/hikoworks/hikogui/blob/main/docs/BON8.md) + - [Reference implementation](https://github.com/hikoworks/hikogui/blob/main/src/hikogui/codec/BON8.hpp) in HikoGUI + +## Serialization + +The library uses the following mapping from JSON values types to BON8 types according to the BON8 specification: + +| JSON value type | value/range | BON8 type | first byte | +|-----------------|----------------------------------------------|-------------------------------|------------| +| null | `null` | null | 0xFA | +| boolean | `true` | true | 0xF9 | +| boolean | `false` | false | 0xF8 | +| number_integer | -9223372036854775808..-2147483649 | int64 | 0x8D | +| number_integer | -2147483648..-33818507 | int32 | 0x8C | +| number_integer | -33818506..-264075 | 4-byte negative integer | 0xF0..0xF7 | +| number_integer | -264074..-1931 | 3-byte negative integer | 0xE0..0xEF | +| number_integer | -1930..-11 | 2-byte negative integer | 0xC2..0xDF | +| number_integer | -10..-1 | 1-byte negative integer | 0xB8..0xC1 | +| number_integer | 0..39 | 1-byte positive integer | 0x90..0xB7 | +| number_integer | 40..3879 | 2-byte positive integer | 0xC2..0xDF | +| number_integer | 3880..528167 | 3-byte positive integer | 0xE0..0xEF | +| number_integer | 528168..67637031 | 4-byte positive integer | 0xF0..0xF7 | +| number_integer | 67637032..2147483647 | int32 | 0x8C | +| number_integer | 2147483648..9223372036854775807 | int64 | 0x8D | +| number_unsigned | 0..39 | 1-byte positive integer | 0x90..0xB7 | +| number_unsigned | 40..3879 | 2-byte positive integer | 0xC2..0xDF | +| number_unsigned | 3880..528167 | 3-byte positive integer | 0xE0..0xEF | +| number_unsigned | 528168..67637031 | 4-byte positive integer | 0xF0..0xF7 | +| number_unsigned | 67637032..2147483647 | int32 | 0x8C | +| number_unsigned | 2147483648..9223372036854775807 | int64 | 0x8D | +| number_float | `-1.0` | -1.0 | 0xFB | +| number_float | `0.0` | 0.0 | 0xFC | +| number_float | `1.0` | 1.0 | 0xFD | +| number_float | *any other value representable by a float* | binary32 | 0x8E | +| number_float | *any value NOT representable by a float* | binary64 | 0x8F | +| string | *empty* | end of string | 0xFF | +| string | *non-empty* | UTF-8 string | 0x00..0x7F, 0xC2..0xF4 | +| array | *size*: 0..4 | array with count | 0x80..0x84 | +| array | *size*: 5 or more | array (terminated by 0xFE) | 0x85 | +| object | *size*: 0..4 | object with count | 0x86..0x8A | +| object | *size*: 5 or more | object (terminated by 0xFE) | 0x8B | +| binary | *size*: 0..4 | array with count | 0x80..0x84 | +| binary | *size*: 5 or more | array (terminated by 0xFE) | 0x85 | + +An integer that takes 2 to 4 bytes starts with a UTF-8 lead byte (0xC2..0xF7) that is followed by a byte that cannot +continue a UTF-8 character: 0x00..0x7F for positive and 0xC0..0xFF for negative integers. A string is terminated by +0xFF only if it is empty, if another string follows it, or if it is the last value of the message; otherwise, the first +byte of the next value ends it. + +!!! success "Complete mapping" + + Except for the values listed below, any JSON value can be converted to a BON8 value. + + Any BON8 output created by `to_bon8` can be successfully parsed by `from_bon8`. + +!!! warning "Unsupported values" + + The following values can **not** be converted to a BON8 value: + + - unsigned integers above 9223372036854775807, because BON8 has no unsigned 64-bit integer type + ([out_of_range.407](../../home/exceptions.md#jsonexceptionout_of_range407)) + - strings that are not valid UTF-8, because the end of a string is determined from its encoding + ([type_error.316](../../home/exceptions.md#jsonexceptiontype_error316)) + +!!! info "NaN/infinity handling" + + `-0.0`, `Infinity`, and `-Infinity` are serialized as binary32 (type 0x8E, 5 bytes total). `NaN` is serialized as + the binary32 value 0x7F800001 that the specification recommends. This is in contrast to the + [dump](../../api/basic_json/dump.md) function which serializes NaN or Infinity to `null`. + +!!! warning "Binary values" + + BON8 has no binary type. Binary values are serialized as arrays of integers (0..255), so they are read back as + arrays. The subtype is not serialized. + +!!! info "Canonical representation" + + The output follows the specification's canonical representation rules: every value uses the shortest encoding, + floating-point numbers use binary32 whenever that loses no precision, and object keys are sorted by their UTF-8 + code units. There are two exceptions: + + - Strings are not normalized to Unicode Normalization Form C (NFC). + - Object keys are written in the order of the object type, which is sorted for `json`, but not for + [`ordered_json`](../../api/ordered_json.md). + +??? example + + ```cpp + --8<-- "examples/to_bon8.cpp" + ``` + + Output: + + ```c + --8<-- "examples/to_bon8.output" + ``` + +## Deserialization + +The library maps BON8 types to JSON value types as follows: + +| BON8 type | JSON value type | first byte | +|-------------------------------|-----------------|------------------------| +| UTF-8 string | string | 0x00..0x7F | +| array with count | array | 0x80..0x84 | +| array (terminated by 0xFE) | array | 0x85 | +| object with count | object | 0x86..0x8A | +| object (terminated by 0xFE) | object | 0x8B | +| int32 | number_unsigned or number_integer | 0x8C | +| int64 | number_unsigned or number_integer | 0x8D | +| binary32 | number_float | 0x8E | +| binary64 | number_float | 0x8F | +| 1-byte positive integer | number_unsigned | 0x90..0xB7 | +| 1-byte negative integer | number_integer | 0xB8..0xC1 | +| UTF-8 string | string | 0xC2..0xF4, followed by 0x80..0xBF | +| 2- to 4-byte positive integer | number_unsigned | 0xC2..0xF7, followed by 0x00..0x7F | +| 2- to 4-byte negative integer | number_integer | 0xC2..0xF7, followed by 0xC0..0xFF | +| false | `false` | 0xF8 | +| true | `true` | 0xF9 | +| null | `null` | 0xFA | +| -1.0 | number_float | 0xFB | +| 0.0 | number_float | 0xFC | +| 1.0 | number_float | 0xFD | +| empty string | string | 0xFF | + +Non-negative integers are read as number_unsigned, negative integers as number_integer. + +!!! info + + Values that do not use the canonical representation, such as integers with a longer encoding than necessary, + arrays and objects with up to four elements that are terminated by 0xFE, unsorted object keys, or a 0xFF after a + string that would also end without it, are accepted. A second 0xFF is not a terminator but an empty string. + + Strings must be valid UTF-8, and the last string of a message must be terminated by 0xFF. + +!!! info + + Any BON8 output created by `to_bon8` can be successfully parsed by `from_bon8`. + +??? example + + ```cpp + --8<-- "examples/from_bon8.cpp" + ``` + + Output: + + ```json + --8<-- "examples/from_bon8.output" + ``` diff --git a/docs/mkdocs/docs/features/binary_formats/index.md b/docs/mkdocs/docs/features/binary_formats/index.md index ef79e2ef3..2c45995a6 100644 --- a/docs/mkdocs/docs/features/binary_formats/index.md +++ b/docs/mkdocs/docs/features/binary_formats/index.md @@ -4,6 +4,7 @@ Though JSON is a ubiquitous data format, it is not a very compact format suitabl a network. Hence, the library supports - [BJData](bjdata.md) (Binary JData), +- [BON8](bon8.md) (Binary Object Notation 8), - [BSON](bson.md) (Binary JSON), - [CBOR](cbor.md) (Concise Binary Object Representation), - [MessagePack](messagepack.md), and @@ -18,6 +19,7 @@ to efficiently encode JSON values to byte vectors and to decode such vectors. | Format | Serialization | Deserialization | |-------------|-----------------------------------------------|----------------------------------------------| | BJData | complete | complete | +| BON8 | incomplete: no unsigned integers above int64 | complete | | BSON | incomplete: top-level value must be an object | incomplete, but all JSON types are supported | | CBOR | complete | incomplete, but all JSON types are supported | | MessagePack | complete | complete | @@ -28,6 +30,7 @@ to efficiently encode JSON values to byte vectors and to decode such vectors. | Format | Binary values | Binary subtypes | |-------------|---------------|-----------------| | BJData | not supported | not supported | +| BON8 | not supported | not supported | | BSON | supported | supported | | CBOR | supported | supported | | MessagePack | supported | supported | @@ -42,6 +45,7 @@ See [binary values](../binary_values.md) for more information. | BJData | 53.2 % | 91.1 % | 78.1 % | 96.6 % | | BJData (size) | 58.6 % | 92.1 % | 86.7 % | 97.4 % | | BJData (size+type) | 58.6 % | 92.1 % | 86.5 % | 97.4 % | +| BON8 | 50.5 % | 83.8 % | 63.5 % | 87.5 % | | BSON | 85.8 % | 95.2 % | 95.8 % | 106.7 % | | CBOR | 50.5 % | 86.3 % | 68.4 % | 88.0 % | | MessagePack | 50.5 % | 86.0 % | 68.5 % | 87.9 % | diff --git a/docs/mkdocs/docs/features/binary_values.md b/docs/mkdocs/docs/features/binary_values.md index 79629383f..aa3185c2a 100644 --- a/docs/mkdocs/docs/features/binary_values.md +++ b/docs/mkdocs/docs/features/binary_values.md @@ -187,6 +187,41 @@ as an array of uint8 values. The library implements this translation. } ``` +### BON8 + +[BON8](binary_formats/bon8.md) neither supports binary values nor subtypes. The library serializes binary values as an +array of integers. + +??? example + + Code: + + ```cpp + // create a binary value of subtype 42 (will be ignored in BON8) + json j; + j["binary"] = json::binary({0xCA, 0xFE, 0xBA, 0xBE}, 42); + + // convert to BON8 + auto v = json::to_bon8(j); + ``` + + `v` is a `std::vector` with the following 16 elements: + + ```c + 0x87 // object with 1 member + 0x62 0x69 0x6E 0x61 0x72 0x79 // "binary" + 0x84 // array with 4 elements + 0xC3 0x22 0xC3 0x56 0xC3 0x12 0xC3 0x16 // content (each byte as a 2-byte integer) + ``` + + Note that the subtype is lost, and deserializing `v` would yield the following value: + + ```json + { + "binary": [202, 254, 186, 190] + } + ``` + ### BSON [BSON](binary_formats/bson.md) supports binary values and subtypes. If a subtype is given, it is used and added as an diff --git a/docs/mkdocs/docs/features/index.md b/docs/mkdocs/docs/features/index.md index 0bbb2dadd..33403a58c 100644 --- a/docs/mkdocs/docs/features/index.md +++ b/docs/mkdocs/docs/features/index.md @@ -35,8 +35,8 @@ C++ types, and finally serialize it again. - [Serialization](serialization.md) — turn a value back into JSON text with [`dump`](../api/basic_json/dump.md), including pretty-printing and handling of non-ASCII and invalid UTF-8. - [Binary formats](binary_formats/index.md) — encode values more compactly as - [BJData](binary_formats/bjdata.md), [BSON](binary_formats/bson.md), [CBOR](binary_formats/cbor.md), - [MessagePack](binary_formats/messagepack.md), or [UBJSON](binary_formats/ubjson.md). + [BJData](binary_formats/bjdata.md), [BON8](binary_formats/bon8.md), [BSON](binary_formats/bson.md), + [CBOR](binary_formats/cbor.md), [MessagePack](binary_formats/messagepack.md), or [UBJSON](binary_formats/ubjson.md). - [Binary values](binary_values.md) — store and exchange raw byte sequences. ## How values are stored and configured diff --git a/docs/mkdocs/docs/features/serialization.md b/docs/mkdocs/docs/features/serialization.md index a875fcdde..5b916e6c8 100644 --- a/docs/mkdocs/docs/features/serialization.md +++ b/docs/mkdocs/docs/features/serialization.md @@ -117,7 +117,7 @@ For the [{fmt}](https://github.com/fmtlib/fmt) library, the library ships a ## Serializing to other formats Besides JSON text, a value can also be serialized to the more compact [binary formats](binary_formats/index.md) -(BJData, BSON, CBOR, MessagePack, UBJSON). +(BJData, BON8, BSON, CBOR, MessagePack, UBJSON). ## See also diff --git a/docs/mkdocs/docs/features/types/template_parameters.md b/docs/mkdocs/docs/features/types/template_parameters.md index 2a670f81c..8ed23e156 100644 --- a/docs/mkdocs/docs/features/types/template_parameters.md +++ b/docs/mkdocs/docs/features/types/template_parameters.md @@ -547,7 +547,7 @@ Grisu2 algorithm, which produces the shortest representation that round-trips. O ### Required for the binary formats `NumberFloatType` must be `#!cpp float` or `#!cpp double`. The writers for -[CBOR, MessagePack, UBJSON, BJData, and BSON](../binary_formats/index.md) map a floating-point value onto an IEEE 754 +[CBOR, MessagePack, UBJSON, BJData, BON8, and BSON](../binary_formats/index.md) map a floating-point value onto an IEEE 754 binary32 or binary64 field and have no encoding for `#!cpp long double`. ### Compatible types diff --git a/docs/mkdocs/includes/glossary.md b/docs/mkdocs/includes/glossary.md index afc7ebc6a..9f2c6329d 100644 --- a/docs/mkdocs/includes/glossary.md +++ b/docs/mkdocs/includes/glossary.md @@ -5,6 +5,7 @@ *[ASCII]: American Standard Code for Information Interchange *[BDFL]: Benevolent Dictator for Life *[BJData]: Binary JData +*[BON8]: Binary Object Notation 8 *[BSON]: Binary JSON *[CBOR]: Concise Binary Object Representation *[CC0]: Creative Commons Zero diff --git a/docs/mkdocs/mkdocs.yml b/docs/mkdocs/mkdocs.yml index a05ce2dff..70537bd27 100644 --- a/docs/mkdocs/mkdocs.yml +++ b/docs/mkdocs/mkdocs.yml @@ -63,6 +63,7 @@ nav: - Binary Formats: - features/binary_formats/index.md - features/binary_formats/bjdata.md + - features/binary_formats/bon8.md - features/binary_formats/bson.md - features/binary_formats/cbor.md - features/binary_formats/messagepack.md @@ -142,6 +143,7 @@ nav: - 'flatten': api/basic_json/flatten.md - 'format_as': api/basic_json/format_as.md - 'from_bjdata': api/basic_json/from_bjdata.md + - 'from_bon8': api/basic_json/from_bon8.md - 'from_bson': api/basic_json/from_bson.md - 'from_cbor': api/basic_json/from_cbor.md - 'from_msgpack': api/basic_json/from_msgpack.md @@ -213,6 +215,7 @@ nav: - 'swap': api/basic_json/swap.md - 'std::swap<basic_json>': api/basic_json/std_swap.md - 'to_bjdata': api/basic_json/to_bjdata.md + - 'to_bon8': api/basic_json/to_bon8.md - 'to_bson': api/basic_json/to_bson.md - 'to_cbor': api/basic_json/to_cbor.md - 'to_msgpack': api/basic_json/to_msgpack.md @@ -412,7 +415,7 @@ plugins: markdown_description: > JSON for Modern C++ is a C++11 header-only library implementing a JSON value type with an STL-like API, JSON Pointer/Patch, CBOR/MessagePack/ - BSON/UBJSON/BJData binary format support, and a SAX-style parser interface. + BSON/UBJSON/BJData/BON8 binary format support, and a SAX-style parser interface. sections: Home: - index.md diff --git a/include/nlohmann/detail/input/binary_reader.hpp b/include/nlohmann/detail/input/binary_reader.hpp index b6c219fc1..b0675c626 100644 --- a/include/nlohmann/detail/input/binary_reader.hpp +++ b/include/nlohmann/detail/input/binary_reader.hpp @@ -28,6 +28,7 @@ #include #include #include +#include #include #include #include @@ -84,7 +85,7 @@ JSON_INLINE_VARIABLE constexpr std::size_t max_valueless_container_size = 1 << 2 /////////////////// /*! -@brief deserialization of CBOR, MessagePack, and UBJSON values +@brief deserialization of BJData, BON8, BSON, CBOR, MessagePack, and UBJSON values */ template> class binary_reader @@ -98,6 +99,11 @@ class binary_reader using char_type = typename InputAdapterType::char_type; using char_int_type = typename char_traits::int_type; + /// whether the input is a contiguous block of bytes that can be inspected + /// and consumed in bulk (as in the lexer); used by @ref get_bon8_string_bulk + static constexpr bool bulk_scan = + input_adapter_supports_bulk_scan(is_detected {}); + public: /*! @brief create a binary reader @@ -132,6 +138,7 @@ class binary_reader { sax = sax_; container_stack.clear(); + bon8_pushback_size = 0; bool result = false; switch (format) @@ -153,6 +160,10 @@ class binary_reader result = parse_ubjson_internal(); break; + case input_format_t::bon8: + result = parse_bon8_internal(); + break; + case input_format_t::json: // LCOV_EXCL_LINE default: // LCOV_EXCL_LINE JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE @@ -165,6 +176,11 @@ class binary_reader { get_ignore_noop(); } + else if (input_format == input_format_t::bon8) + { + // a string that ends a container hands back the byte after it + get_bon8(); + } else { get(); @@ -396,6 +412,11 @@ class binary_reader */ bool get_bson_cstr(string_t& result) { + if (get_bson_cstr_bulk(result, std::integral_constant {})) + { + return true; + } + auto out = std::back_inserter(result); while (true) { @@ -412,6 +433,46 @@ class binary_reader } } + /*! + @brief read a C-style string from contiguous input in one step + + @param[in,out] result the string to append to + @return whether the string was read; if the input has no \x00-byte, nothing + is read, and @ref get_bson_cstr reports the end of the input + */ + bool get_bson_cstr_bulk(string_t& result, std::true_type /*bulk*/) + { + const std::size_t remaining = ia.bulk_remaining(); + if (remaining == 0) + { + return false; + } + const auto* const data = reinterpret_cast(ia.bulk_data()); + // a plain loop rather than std::memchr: most keys are short (array + // indices are keys, too), and the call would cost more than it saves + std::size_t length = 0; + while (length < remaining && data[length] != 0x00) + { + ++length; + } + if (length == remaining) + { + return false; + } + result.append(reinterpret_cast(data), length); + // consume the string and its \x00-byte, as the byte-wise path does + ia.bulk_skip(length + 1); + chars_read += length + 1; + current = 0x00; + return true; + } + + /// input that is not contiguous: C-style strings are read byte by byte + bool get_bson_cstr_bulk(string_t& /*result*/, std::false_type /*bulk*/) const noexcept + { + return false; + } + /*! @brief Parses a zero-terminated string of length @a len from the BSON input. @@ -3199,6 +3260,549 @@ class binary_reader } } + ////////// + // BON8 // + ////////// + + /*! + @brief get the next byte of a BON8 value + + A BON8 string has no length prefix and no mandatory terminator: it ends at + the first byte that cannot continue it, which is already the first byte (or, + for an integer that begins with a UTF-8 lead byte, the first two bytes) of + whatever follows. The string reader hands those bytes back with + @ref unget_bon8, and every BON8 read goes through this function so that + they are seen again. + + @return character read from the input + */ + char_int_type get_bon8() + { + if (bon8_pushback_size != 0) + { + ++chars_read; + return current = bon8_pushback[--bon8_pushback_size]; + } + return get(); + } + + /*! + @brief hand a byte back so that the next @ref get_bon8 returns it again + + @param[in] c the byte to hand back; bytes handed back are returned in + reverse order + */ + void unget_bon8(const char_int_type c) + { + // At most two bytes are ever handed back: a byte is only handed back + // right after it was read with get_bon8(), and the only place that + // hands back two bytes (a lead byte and the byte after it) read both + // of them in a row, which emptied the buffer first. This is an + // invariant of the reader rather than a property of the input, so + // an assertion suffices (the fuzzers are built with assertions). + JSON_ASSERT(bon8_pushback_size < bon8_pushback.size()); + bon8_pushback[bon8_pushback_size++] = c; + --chars_read; + } + + /*! + @param[in] c a byte + @return whether @a c is a UTF-8 continuation byte (0x80..0xBF) + */ + static constexpr bool is_bon8_continuation(const char_int_type c) noexcept + { + return 0x80 <= c && c <= 0xBF; + } + + /*! + @brief report a parse error at the last read byte + + @param[in] detail a detailed error message + @param[in] context further context information + @return false + */ + bool bon8_error(const std::string& detail, const char* context) + { + auto last_token = get_token_string(); + return sax->parse_error(chars_read, last_token, parse_error::create(112, chars_read, + exception_message(input_format_t::bon8, concat(detail, ": 0x", last_token), context), nullptr)); + } + + /*! + @brief read a BON8 value and everything nested inside it + + Reads values until the one that was begun here is complete, resuming the + enclosing container after each element, so that the nesting depth of the + input costs heap rather than native stack (see #5104). + + @return whether reading the value succeeded + */ + bool parse_bon8_internal() + { + // the key currently being read; hoisted out of the loop so that its + // capacity is reused across elements and across nesting levels + string_t key; + + while (true) + { + if (!container_stack.empty()) + { + // a copy, not a reference: it must stay valid across the + // pop_back() below, which destroys the container_stack element + // it would otherwise alias + const container_frame top = container_stack.back(); + bool at_end = false; + + if (top.remaining != npos) + { + // counted container (0x80..0x84, 0x86..0x8A): it ends once + // its elements have been read + at_end = (top.remaining == 0); + if (!at_end) + { + // claim the element about to be read + --container_stack.back().remaining; + } + } + else + { + // container 0x85 or 0x8B: it ends at an end-of-container + // marker (0xFE); any other byte begins the next element + at_end = (get_bon8() == 0xFE); + if (!at_end) + { + unget_bon8(current); + } + } + + if (at_end) + { + container_stack.pop_back(); + if (JSON_HEDLEY_UNLIKELY(top.is_object ? !sax->end_object() : !sax->end_array())) + { + return false; + } + // the value begun here is complete once its container is + if (container_stack.empty()) + { + return true; + } + continue; + } + + if (top.is_object) + { + key.clear(); + if (JSON_HEDLEY_UNLIKELY(!get_bon8_key(key) || !sax->key(key))) + { + return false; + } + } + } + + if (JSON_HEDLEY_UNLIKELY(!parse_bon8_value())) + { + return false; + } + + // a value that opened a container left it on the stack; one that + // did not, and that was not inside a container, was the whole value + if (container_stack.empty()) + { + return true; + } + } + } + + /*! + @brief read one BON8 value + + Reads a single value and passes it to the SAX parser. A value that begins + a container is not read to its end: the container is opened with + @ref enter_container and its elements are read by + @ref parse_bon8_internal, so that nesting does not consume native stack. + + @return whether reading the value succeeded + */ + bool parse_bon8_value() + { + const auto byte = get_bon8(); + + if (byte == char_traits::eof()) + { + return unexpect_eof(input_format_t::bon8, "value"); + } + + // string: ASCII character + if (byte <= 0x7F) + { + string_t s; + unget_bon8(byte); + return get_bon8_string(s) && sax->string(s); + } + + // array with 0..4 elements + if (byte <= 0x84) + { + return enter_array(static_cast(byte - 0x80)); + } + + // array terminated by 0xFE + if (byte == 0x85) + { + return enter_array(npos); + } + + // object with 0..4 members + if (byte <= 0x8A) + { + return enter_object(static_cast(byte - 0x86)); + } + + switch (byte) + { + case 0x8B: // object terminated by 0xFE + return enter_object(npos); + + case 0x8C: // int32 + { + std::int32_t number{}; + return get_number(input_format_t::bon8, number) && emit_bon8_integer(number); + } + + case 0x8D: // int64 + { + std::int64_t number{}; + return get_number(input_format_t::bon8, number) && emit_bon8_integer(number); + } + + case 0x8E: // binary32 + { + float number{}; + return get_number(input_format_t::bon8, number) && sax->number_float(static_cast(number), ""); + } + + case 0x8F: // binary64 + { + double number{}; + return get_number(input_format_t::bon8, number) && sax->number_float(static_cast(number), ""); + } + + case 0xF8: + return sax->boolean(false); + + case 0xF9: + return sax->boolean(true); + + case 0xFA: + return sax->null(); + + case 0xFB: + return sax->number_float(static_cast(-1.0), ""); + + case 0xFC: + return sax->number_float(static_cast(0.0), ""); + + case 0xFD: + return sax->number_float(static_cast(1.0), ""); + + case 0xFF: // empty string + { + string_t s; + return sax->string(s); + } + + default: + break; + } + + // integer 0..39 + if (byte <= 0xB7) + { + return sax->number_unsigned(static_cast(byte - 0x90)); + } + + // integer -1..-10 + if (byte <= 0xC1) + { + return sax->number_integer(-1 - static_cast(byte - 0xB8)); + } + + // 0xC2..0xF7: a UTF-8 lead byte begins a string if a continuation + // byte follows and an integer otherwise + if (byte <= 0xF7) + { + const auto second = get_bon8(); + if (is_bon8_continuation(second)) + { + string_t s; + unget_bon8(second); + unget_bon8(byte); + return get_bon8_string(s) && sax->string(s); + } + return get_bon8_integer(byte, second); + } + + // 0xFE: end of container where a value is expected + return bon8_error("invalid byte", "value"); + } + + /*! + @brief pass an integer to the SAX parser + + Non-negative integers are passed as unsigned, negative integers as signed + numbers, like the other binary formats do. + + @param[in] number the integer + @return whether the SAX parser accepted the value + */ + bool emit_bon8_integer(const std::int64_t number) + { + if (number >= 0) + { + return sax->number_unsigned(static_cast(number)); + } + return sax->number_integer(static_cast(number)); + } + + /*! + @brief read an integer encoded in 2..4 bytes + + The first byte is a UTF-8 lead byte (0xC2..0xF7) that is followed by a + byte that is not a continuation byte: 0x00..0x7F for positive and + 0xC0..0xFF for negative integers. The lead byte's low bits and the second + byte's low 7 (positive) or 6 (negative) bits are the most significant bits + of the value; 3- and 4-byte integers add one or two full bytes. Each range + starts where the shorter one ends, so no value has two encodings of the + same length. + + @param[in] lead the first byte (0xC2..0xF7) + @param[in] second the second byte + @return whether reading the integer succeeded + */ + bool get_bon8_integer(const char_int_type lead, const char_int_type second) + { + if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format_t::bon8, "number"))) + { + return false; + } + + const bool negative = second >= 0xC0; + auto value = static_cast(negative ? (second & 0x3F) : second); + std::int64_t offset = 0; + int extra_bytes = 0; + + if (lead <= 0xDF) + { + value |= static_cast(lead - 0xC2) << (negative ? 6 : 7); + offset = negative ? 11 : 40; + } + else if (lead <= 0xEF) + { + value |= static_cast(lead & 0x0F) << (negative ? 6 : 7); + offset = negative ? 1931 : 3880; + extra_bytes = 1; + } + else + { + value |= static_cast(lead & 0x07) << (negative ? 6 : 7); + offset = negative ? 264075 : 528168; + extra_bytes = 2; + } + + for (int i = 0; i < extra_bytes; ++i) + { + if (JSON_HEDLEY_UNLIKELY(get_bon8() == char_traits::eof())) + { + return unexpect_eof(input_format_t::bon8, "number"); + } + value = (value << 8) | static_cast(current); + } + + return negative ? sax->number_integer(static_cast(-(value + offset))) + : sax->number_unsigned(static_cast(value + offset)); + } + + /*! + @brief read an object key + + A key must be a string, so its first byte must be an ASCII character, a + UTF-8 lead byte followed by a continuation byte, or 0xFF (empty string). + + @param[out] result the key + @return whether reading the key succeeded + */ + bool get_bon8_key(string_t& result) + { + const auto byte = get_bon8(); + + if (byte == char_traits::eof()) + { + return unexpect_eof(input_format_t::bon8, "key"); + } + + if (byte == 0xFF) + { + return true; + } + + if (byte <= 0x7F) + { + unget_bon8(byte); + return get_bon8_string(result); + } + + if (0xC2 <= byte && byte <= 0xF7) + { + const auto second = get_bon8(); + unget_bon8(second); + if (is_bon8_continuation(second)) + { + unget_bon8(byte); + return get_bon8_string(result); + } + // an integer: report its first byte rather than the one after it + current = byte; + } + + return bon8_error("expected a string; last byte", "key"); + } + + /*! + @brief append the run of valid UTF-8 at the read position to a string + + For contiguous input, the ASCII characters and complete well-formed UTF-8 + sequences at the read position are appended to @a result in one step. The + byte that stops the run (an end-of-string marker, the first byte of the + next value, or an ill-formed byte) is left for @ref get_bon8_string, so + that strings end and errors are reported exactly as without this step. + + @param[in,out] result the string to append to + */ + void get_bon8_string_bulk(string_t& result, std::true_type /*bulk*/) + { + // bytes handed back must be read through get_bon8() first + if (bon8_pushback_size != 0) + { + return; + } + const std::size_t remaining = ia.bulk_remaining(); + if (remaining == 0) + { + return; + } + const auto* const data = reinterpret_cast(ia.bulk_data()); + const std::size_t length = valid_utf8_prefix(data, remaining); + if (length != 0) + { + result.append(reinterpret_cast(data), length); + ia.bulk_skip(length); + chars_read += length; + } + } + + /// input that is not contiguous: strings are read byte by byte + void get_bon8_string_bulk(string_t& /*result*/, std::false_type /*bulk*/) const noexcept {} + + /*! + @brief read a string + + Reads UTF-8 characters until an end-of-string marker (0xFF), which is + consumed, or a byte that cannot continue the string, which is handed back + to be read as the start of the next value. The string must be valid UTF-8, + and it must not end at the end of the input: the last string of a message + is always terminated by 0xFF. + + @param[out] result the string + @return whether reading the string succeeded + */ + bool get_bon8_string(string_t& result) + { + while (true) + { + get_bon8_string_bulk(result, std::integral_constant {}); + + const auto byte = get_bon8(); + + if (byte == char_traits::eof()) + { + return unexpect_eof(input_format_t::bon8, "string"); + } + + // end of string + if (byte == 0xFF) + { + return true; + } + + // ASCII character + if (byte <= 0x7F) + { + result.push_back(static_cast(byte)); + continue; + } + + // a byte that cannot begin a character ends the string and begins + // the next value + if (byte < 0xC2 || byte > 0xF7) + { + unget_bon8(byte); + return true; + } + + // a lead byte ends the string if no continuation byte follows: it + // is then the first byte of an integer + const auto second = get_bon8(); + if (!is_bon8_continuation(second)) + { + unget_bon8(second); + unget_bon8(byte); + return true; + } + + // the valid range of the second byte excludes overlong forms, + // surrogates, and code points above U+10FFFF + // (RFC 3629, section 4) + int continuation_bytes = 0; + bool valid_second = true; + if (byte <= 0xDF) + { + continuation_bytes = 1; + } + else if (byte <= 0xEF) + { + continuation_bytes = 2; + valid_second = (byte != 0xE0 || second >= 0xA0) && (byte != 0xED || second <= 0x9F); + } + else + { + continuation_bytes = 3; + valid_second = byte <= 0xF4 && (byte != 0xF0 || second >= 0x90) && (byte != 0xF4 || second <= 0x8F); + } + + if (JSON_HEDLEY_UNLIKELY(!valid_second)) + { + return bon8_error("invalid UTF-8 byte", "string"); + } + + result.push_back(static_cast(byte)); + result.push_back(static_cast(second)); + + for (int i = 1; i < continuation_bytes; ++i) + { + if (JSON_HEDLEY_UNLIKELY(get_bon8() == char_traits::eof())) + { + return unexpect_eof(input_format_t::bon8, "string"); + } + if (JSON_HEDLEY_UNLIKELY(!is_bon8_continuation(current))) + { + return bon8_error("invalid UTF-8 byte", "string"); + } + result.push_back(static_cast(current)); + } + } + } + /////////////////////// // Utility functions // /////////////////////// @@ -3497,6 +4101,10 @@ class binary_reader error_msg += "BJData"; break; + case input_format_t::bon8: + error_msg += "BON8"; + break; + case input_format_t::json: // LCOV_EXCL_LINE default: // LCOV_EXCL_LINE JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE @@ -3529,6 +4137,11 @@ class binary_reader /// the containers that have been opened and not closed yet; see @ref container_frame std::vector container_stack{}; + /// BON8: bytes read past the end of a string, returned again by @ref get_bon8 + std::array bon8_pushback{{}}; + /// BON8: number of bytes in @ref bon8_pushback + std::size_t bon8_pushback_size = 0; + // excluded markers in bjdata optimized type #define JSON_BINARY_READER_MAKE_BJD_OPTIMIZED_TYPE_MARKERS_ \ make_array('F', 'H', 'N', 'S', 'T', 'Z', '[', '{') diff --git a/include/nlohmann/detail/input/input_adapters.hpp b/include/nlohmann/detail/input/input_adapters.hpp index 317b38806..e174775c5 100644 --- a/include/nlohmann/detail/input/input_adapters.hpp +++ b/include/nlohmann/detail/input/input_adapters.hpp @@ -34,7 +34,7 @@ namespace detail { /// the supported input formats -enum class input_format_t { json, cbor, msgpack, ubjson, bson, bjdata }; +enum class input_format_t { json, cbor, msgpack, ubjson, bson, bjdata, bon8 }; //////////////////// // input adapters // diff --git a/include/nlohmann/detail/input/string_scan.hpp b/include/nlohmann/detail/input/string_scan.hpp index 6af0e6c5d..8a6f03122 100644 --- a/include/nlohmann/detail/input/string_scan.hpp +++ b/include/nlohmann/detail/input/string_scan.hpp @@ -201,6 +201,43 @@ inline std::size_t validate_one_utf8(const unsigned char* data, std::size_t avai return 0; // invalid, incomplete, or must be diagnosed by the byte path } +// Return the length of the longest prefix of [data, data+n) that consists of +// ASCII characters and complete well-formed UTF-8 sequences; n if all of it is +// valid UTF-8. Unlike scalar_string_bulk_run(), quotes, escapes, and control +// characters are ordinary characters here. ASCII is skipped 8 bytes at a time. +inline std::size_t valid_utf8_prefix(const unsigned char* data, std::size_t n) noexcept +{ + constexpr std::uint64_t high = 0x8080808080808080ull; + std::size_t pos = 0; + while (pos < n) + { + if (pos + 8 <= n) + { + std::uint64_t word = 0; + std::memcpy(&word, data + pos, sizeof(word)); + if ((word & high) == 0) + { + pos += 8; + continue; + } + } + + if (data[pos] < 0x80u) + { + ++pos; + continue; + } + + const std::size_t seq = validate_one_utf8(data + pos, n - pos); + if (seq == 0) + { + break; // ill-formed or truncated + } + pos += seq; + } + return pos; +} + // Scalar (C++11) computation of the bulk run length: the number of leading // bytes in [data, data+n) that are ordinary ASCII or complete well-formed UTF-8 // sequences, stopping before the first byte that needs individual handling (the diff --git a/include/nlohmann/detail/output/binary_writer.hpp b/include/nlohmann/detail/output/binary_writer.hpp index 5b11a3b2f..9da4269a8 100644 --- a/include/nlohmann/detail/output/binary_writer.hpp +++ b/include/nlohmann/detail/output/binary_writer.hpp @@ -25,6 +25,7 @@ #endif #include +#include #include #include #include @@ -76,7 +77,7 @@ std::size_t binary_reserve_hint(const BasicJsonType& j) } /*! -@brief serialization to CBOR and MessagePack values +@brief serialization to BJData, BON8, BSON, CBOR, MessagePack, and UBJSON values */ template> class binary_writer @@ -873,6 +874,21 @@ class binary_writer } } + /*! + @param[in] j JSON value to serialize + */ + void write_bon8(const BasicJsonType& j) + { + bool string_open = false; + write_bon8_value(j, string_open); + + // the last string of a message must be terminated + if (string_open) + { + oa.write_character(to_char_type(0xFF)); + } + } + private: ////////// // BSON // @@ -1431,6 +1447,28 @@ class binary_writer return to_char_type(0xCB); // float 64 } + /// @return the BON8 type marker for binary32 (float) or binary64 (double) + template + static constexpr CharType get_bon8_float_prefix() + { + return to_char_type(std::is_same::value ? 0x8E : 0x8F); + } + + /// @return the type marker for a FloatType value in @a format (CBOR, MessagePack, or BON8) + template + static CharType get_compact_float_prefix(const detail::input_format_t format) + { + if (format == detail::input_format_t::cbor) + { + return get_cbor_float_prefix(FloatType{}); + } + if (format == detail::input_format_t::bon8) + { + return get_bon8_float_prefix(); + } + return get_msgpack_float_prefix(FloatType{}); + } + //////////// // UBJSON // //////////// @@ -2050,6 +2088,322 @@ class binary_writer return false; } + ////////// + // BON8 // + ////////// + + /*! + @brief write a BON8 value + + A string is written without length or terminator: it ends at the first + byte that cannot continue it, which is the first byte of any non-string + value and of the end-of-container marker 0xFE. It only needs an explicit + end-of-string marker (0xFF) when it is empty, when another string follows, + or when it is the last thing in the message. + + @param[in] j JSON value to serialize + @param[in,out] string_open whether the output ends with a non-empty + string that has not been terminated with 0xFF + */ + void write_bon8_value(const BasicJsonType& j, bool& string_open) + { + switch (j.type()) + { + case value_t::null: + { + write_bon8_marker(0xFA, string_open); + break; + } + + case value_t::boolean: + { + write_bon8_marker(j.m_data.m_value.boolean ? 0xF9 : 0xF8, string_open); + break; + } + + case value_t::number_unsigned: + { + if (j.m_data.m_value.number_unsigned > static_cast((std::numeric_limits::max)())) + { + JSON_THROW(out_of_range::create(407, concat("integer number ", std::to_string(j.m_data.m_value.number_unsigned), " cannot be represented by BON8 as it does not fit int64"), &j)); + } + write_bon8_integer(static_cast(j.m_data.m_value.number_unsigned)); + string_open = false; + break; + } + + case value_t::number_integer: + { + write_bon8_integer(static_cast(j.m_data.m_value.number_integer)); + string_open = false; + break; + } + + case value_t::number_float: + { + write_bon8_float(j.m_data.m_value.number_float); + string_open = false; + break; + } + + case value_t::string: + { + write_bon8_string(*j.m_data.m_value.string, string_open, j); + break; + } + + case value_t::array: + { + const auto N = j.m_data.m_value.array->size(); + // 0x80..0x84: array with 0..4 elements; 0x85: array ended by 0xFE + write_bon8_marker(static_cast(N <= 4 ? 0x80 + N : 0x85), string_open); + + for (const auto& el : *j.m_data.m_value.array) + { + write_bon8_value(el, string_open); + } + + if (N > 4) + { + write_bon8_marker(0xFE, string_open); + } + break; + } + + case value_t::object: + { + const auto N = j.m_data.m_value.object->size(); + // 0x86..0x8A: object with 0..4 members; 0x8B: object ended by 0xFE + write_bon8_marker(static_cast(N <= 4 ? 0x86 + N : 0x8B), string_open); + + for (const auto& el : *j.m_data.m_value.object) + { + write_bon8_string(el.first, string_open, j); + write_bon8_value(el.second, string_open); + } + + if (N > 4) + { + write_bon8_marker(0xFE, string_open); + } + break; + } + + case value_t::binary: + { + // BON8 has no binary type: write the bytes as an array of + // integers, like UBJSON and BJData do + const auto N = j.m_data.m_value.binary->size(); + write_bon8_marker(static_cast(N <= 4 ? 0x80 + N : 0x85), string_open); + + for (std::size_t i = 0; i < N; ++i) + { + // the cast is needed for binary types whose value type + // is not an integer (e.g., std::byte) + write_bon8_integer(static_cast(j.m_data.m_value.binary->data()[i])); + } + + if (N > 4) + { + write_bon8_marker(0xFE, string_open); + } + break; + } + + case value_t::discarded: + default: + break; + } + } + + /*! + @brief write a single byte that is not part of a string + + @param[in] marker the byte to write + @param[out] string_open set to false, because the output no longer ends + with a string; see @ref write_bon8_value + */ + void write_bon8_marker(const std::uint8_t marker, bool& string_open) + { + oa.write_character(to_char_type(marker)); + string_open = false; + } + + /*! + @brief write a string + + @param[in] s the string to write + @param[in,out] string_open see @ref write_bon8_value + @param[in] context the value the string belongs to (for diagnostics) + + @throw type_error.316 if @a s is not valid UTF-8, because the end of a + string is determined from its encoding + */ + void write_bon8_string(const string_t& s, bool& string_open, const BasicJsonType& context) + { + check_bon8_utf8(s, context); + + // a string that follows another string terminates it + if (string_open) + { + oa.write_character(to_char_type(0xFF)); + } + + if (s.empty()) + { + // the empty string is just the end-of-string marker + oa.write_character(to_char_type(0xFF)); + string_open = false; + } + else + { + oa.write_characters(reinterpret_cast(s.data()), s.size()); + string_open = true; + } + } + + /*! + @brief check that a string is valid UTF-8 (RFC 3629) + + @param[in] s the string to check + @param[in] context the value the string belongs to (for diagnostics) + + @throw type_error.316 if @a s is not valid UTF-8; the message names the + first byte of the first invalid or incomplete sequence + */ + static void check_bon8_utf8(const string_t& s, const BasicJsonType& context) + { + static_cast(context); // only used when exceptions are enabled + const auto* data = reinterpret_cast(s.data()); + const std::size_t valid = valid_utf8_prefix(data, s.size()); + if (JSON_HEDLEY_UNLIKELY(valid != s.size())) + { + JSON_THROW(type_error::create(316, concat("invalid UTF-8 byte at index ", std::to_string(valid), ": 0x", hex_byte(data[valid])), &context)); + } + } + + /// @return a byte as two uppercase hexadecimal digits + static std::string hex_byte(const std::uint8_t byte) + { + std::string result = "00"; + constexpr const char* nibble_to_hex = "0123456789ABCDEF"; + result[0] = nibble_to_hex[byte / 16]; + result[1] = nibble_to_hex[byte % 16]; + return result; + } + + /*! + @brief write an integer in the shortest encoding + + Integers from -10 to 39 take one byte. Up to -33818506 and 67637031, an + integer takes 2 to 4 bytes that begin with a UTF-8 lead byte (0xC2..0xF7) + followed by a byte that is not a continuation byte: 0x00..0x7F for + positive and 0xC0..0xFF for negative integers. Each range starts where the + shorter one ends. Larger integers are written as int32 (0x8C) or int64 + (0x8D) in big-endian byte order. + + @param[in] value the integer to write + */ + void write_bon8_integer(std::int64_t value) + { + if (value < (std::numeric_limits::min)() || value > (std::numeric_limits::max)()) + { + oa.write_character(to_char_type(0x8D)); + write_number(value); + } + else if (value < -33818506 || value > 67637031) + { + oa.write_character(to_char_type(0x8C)); + write_number(static_cast(value)); + } + else if (value <= -264075) + { + value = -(value + 264075); + write_bon8_bytes(0xF0 + ((value >> 22) & 0x07), 0xC0 + ((value >> 16) & 0x3F), value >> 8, value); + } + else if (value <= -1931) + { + value = -(value + 1931); + write_bon8_bytes(0xE0 + ((value >> 14) & 0x0F), 0xC0 + ((value >> 8) & 0x3F), value); + } + else if (value <= -11) + { + value = -(value + 11); + write_bon8_bytes(0xC2 + ((value >> 6) & 0x1F), 0xC0 + (value & 0x3F)); + } + else if (value <= -1) + { + write_bon8_bytes(0xB8 - (value + 1)); + } + else if (value <= 39) + { + write_bon8_bytes(0x90 + value); + } + else if (value <= 3879) + { + value -= 40; + write_bon8_bytes(0xC2 + ((value >> 7) & 0x1F), value & 0x7F); + } + else if (value <= 528167) + { + value -= 3880; + write_bon8_bytes(0xE0 + ((value >> 15) & 0x0F), (value >> 8) & 0x7F, value); + } + else + { + value -= 528168; + write_bon8_bytes(0xF0 + ((value >> 23) & 0x07), (value >> 16) & 0x7F, value >> 8, value); + } + } + + /// write the low byte of each argument + template + void write_bon8_bytes(const Bytes... bytes) + { + const std::array buffer{{to_char_type(static_cast(bytes & 0xFF))...}}; + oa.write_characters(buffer.data(), buffer.size()); + } + + /*! + @brief write a floating-point number + + -1.0, +0.0, and 1.0 take one byte. Other numbers are written as binary32 + (0x8E) if that loses no precision, and as binary64 (0x8F) otherwise; -0.0, + infinities, and NaN are always written as binary32, NaN as 0x7F800001. + + @param[in] n the number to write + */ + void write_bon8_float(const number_float_t n) + { +#ifdef __GNUC__ + JSON_HEDLEY_DIAGNOSTIC_PUSH + JSON_HEDLEY_PRAGMA(GCC diagnostic ignored "-Wfloat-equal") +#endif + if (n == static_cast(-1)) + { + oa.write_character(to_char_type(0xFB)); + } + else if (n == static_cast(0) && !std::signbit(n)) + { + oa.write_character(to_char_type(0xFC)); + } + else if (n == static_cast(1)) + { + oa.write_character(to_char_type(0xFD)); + } + else if (std::isnan(n)) + { + write_bon8_bytes(0x8E, 0x7F, 0x80, 0x00, 0x01); + } + else + { + write_compact_float(n, detail::input_format_t::bon8); + } +#ifdef __GNUC__ + JSON_HEDLEY_DIAGNOSTIC_POP +#endif + } + /////////////////////// // Utility functions // /////////////////////// @@ -2184,16 +2538,12 @@ class binary_writer static_cast(n) <= static_cast((std::numeric_limits::max)()) && static_cast(static_cast(n)) == static_cast(n)))) { - oa.write_character(format == detail::input_format_t::cbor - ? get_cbor_float_prefix(static_cast(n)) - : get_msgpack_float_prefix(static_cast(n))); + oa.write_character(get_compact_float_prefix(format)); write_number(static_cast(n)); } else { - oa.write_character(format == detail::input_format_t::cbor - ? get_cbor_float_prefix(n) - : get_msgpack_float_prefix(n)); + oa.write_character(get_compact_float_prefix(format)); write_number(n); } #ifdef __GNUC__ diff --git a/include/nlohmann/json.hpp b/include/nlohmann/json.hpp index abd461d6c..500fcddf2 100644 --- a/include/nlohmann/json.hpp +++ b/include/nlohmann/json.hpp @@ -5290,6 +5290,30 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec binary_writer(o).write_bson(j); } + /// @brief create a BON8 serialization of a given JSON value + /// @sa https://json.nlohmann.me/api/basic_json/to_bon8/ + static std::vector to_bon8(const basic_json& j) + { + std::vector result; + result.reserve(detail::binary_reserve_hint(j)); + vector_writer(result).write_bon8(j); + return result; + } + + /// @brief create a BON8 serialization of a given JSON value + /// @sa https://json.nlohmann.me/api/basic_json/to_bon8/ + static void to_bon8(const basic_json& j, detail::output_adapter o) + { + binary_writer(o).write_bon8(j); + } + + /// @brief create a BON8 serialization of a given JSON value + /// @sa https://json.nlohmann.me/api/basic_json/to_bon8/ + static void to_bon8(const basic_json& j, detail::output_adapter o) + { + binary_writer(o).write_bon8(j); + } + /// @brief create a JSON value from an input in CBOR format /// @sa https://json.nlohmann.me/api/basic_json/from_cbor/ template @@ -5523,6 +5547,43 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec return result; } + /// @brief create a JSON value from an input in BON8 format + /// @sa https://json.nlohmann.me/api/basic_json/from_bon8/ + template + JSON_HEDLEY_WARN_UNUSED_RESULT + static basic_json from_bon8(InputType&& i, + const bool strict = true, + const bool allow_exceptions = true) + { + basic_json result; + auto ia = detail::input_adapter(std::forward(i)); + detail::json_sax_dom_parser sdp(result, allow_exceptions); + if (!binary_reader(std::move(ia), input_format_t::bon8).sax_parse(input_format_t::bon8, &sdp, strict)) // cppcheck-suppress[accessMoved] + { + result = value_t::discarded; + } + return result; + } + + /// @brief create a JSON value from an input in BON8 format (iterator pair, or iterator+sentinel pair for C++20 ranges support) + /// @sa https://json.nlohmann.me/api/basic_json/from_bon8/ + template::value, int> = 0> + JSON_HEDLEY_WARN_UNUSED_RESULT + static basic_json from_bon8(IteratorType first, SentinelType last, + const bool strict = true, + const bool allow_exceptions = true) + { + basic_json result; + auto ia = detail::input_adapter(std::move(first), std::move(last)); + detail::json_sax_dom_parser sdp(result, allow_exceptions); + if (!binary_reader(std::move(ia), input_format_t::bon8).sax_parse(input_format_t::bon8, &sdp, strict)) // cppcheck-suppress[accessMoved] + { + result = value_t::discarded; + } + return result; + } + /// @brief create a JSON value from an input in BSON format /// @sa https://json.nlohmann.me/api/basic_json/from_bson/ template diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 81ddfb1c4..0e3cae486 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -7578,7 +7578,7 @@ namespace detail { /// the supported input formats -enum class input_format_t { json, cbor, msgpack, ubjson, bson, bjdata }; +enum class input_format_t { json, cbor, msgpack, ubjson, bson, bjdata, bon8 }; //////////////////// // input adapters // @@ -8995,6 +8995,43 @@ inline std::size_t validate_one_utf8(const unsigned char* data, std::size_t avai return 0; // invalid, incomplete, or must be diagnosed by the byte path } +// Return the length of the longest prefix of [data, data+n) that consists of +// ASCII characters and complete well-formed UTF-8 sequences; n if all of it is +// valid UTF-8. Unlike scalar_string_bulk_run(), quotes, escapes, and control +// characters are ordinary characters here. ASCII is skipped 8 bytes at a time. +inline std::size_t valid_utf8_prefix(const unsigned char* data, std::size_t n) noexcept +{ + constexpr std::uint64_t high = 0x8080808080808080ull; + std::size_t pos = 0; + while (pos < n) + { + if (pos + 8 <= n) + { + std::uint64_t word = 0; + std::memcpy(&word, data + pos, sizeof(word)); + if ((word & high) == 0) + { + pos += 8; + continue; + } + } + + if (data[pos] < 0x80u) + { + ++pos; + continue; + } + + const std::size_t seq = validate_one_utf8(data + pos, n - pos); + if (seq == 0) + { + break; // ill-formed or truncated + } + pos += seq; + } + return pos; +} + // Scalar (C++11) computation of the bulk run length: the number of leading // bytes in [data, data+n) that are ordinary ASCII or complete well-formed UTF-8 // sequences, stopping before the first byte that needs individual handling (the @@ -12557,6 +12594,8 @@ NLOHMANN_JSON_NAMESPACE_END // #include +// #include + // #include // #include @@ -12781,7 +12820,7 @@ JSON_INLINE_VARIABLE constexpr std::size_t max_valueless_container_size = 1 << 2 /////////////////// /*! -@brief deserialization of CBOR, MessagePack, and UBJSON values +@brief deserialization of BJData, BON8, BSON, CBOR, MessagePack, and UBJSON values */ template> class binary_reader @@ -12795,6 +12834,11 @@ class binary_reader using char_type = typename InputAdapterType::char_type; using char_int_type = typename char_traits::int_type; + /// whether the input is a contiguous block of bytes that can be inspected + /// and consumed in bulk (as in the lexer); used by @ref get_bon8_string_bulk + static constexpr bool bulk_scan = + input_adapter_supports_bulk_scan(is_detected {}); + public: /*! @brief create a binary reader @@ -12829,6 +12873,7 @@ class binary_reader { sax = sax_; container_stack.clear(); + bon8_pushback_size = 0; bool result = false; switch (format) @@ -12850,6 +12895,10 @@ class binary_reader result = parse_ubjson_internal(); break; + case input_format_t::bon8: + result = parse_bon8_internal(); + break; + case input_format_t::json: // LCOV_EXCL_LINE default: // LCOV_EXCL_LINE JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE @@ -12862,6 +12911,11 @@ class binary_reader { get_ignore_noop(); } + else if (input_format == input_format_t::bon8) + { + // a string that ends a container hands back the byte after it + get_bon8(); + } else { get(); @@ -13093,6 +13147,11 @@ class binary_reader */ bool get_bson_cstr(string_t& result) { + if (get_bson_cstr_bulk(result, std::integral_constant {})) + { + return true; + } + auto out = std::back_inserter(result); while (true) { @@ -13109,6 +13168,46 @@ class binary_reader } } + /*! + @brief read a C-style string from contiguous input in one step + + @param[in,out] result the string to append to + @return whether the string was read; if the input has no \x00-byte, nothing + is read, and @ref get_bson_cstr reports the end of the input + */ + bool get_bson_cstr_bulk(string_t& result, std::true_type /*bulk*/) + { + const std::size_t remaining = ia.bulk_remaining(); + if (remaining == 0) + { + return false; + } + const auto* const data = reinterpret_cast(ia.bulk_data()); + // a plain loop rather than std::memchr: most keys are short (array + // indices are keys, too), and the call would cost more than it saves + std::size_t length = 0; + while (length < remaining && data[length] != 0x00) + { + ++length; + } + if (length == remaining) + { + return false; + } + result.append(reinterpret_cast(data), length); + // consume the string and its \x00-byte, as the byte-wise path does + ia.bulk_skip(length + 1); + chars_read += length + 1; + current = 0x00; + return true; + } + + /// input that is not contiguous: C-style strings are read byte by byte + bool get_bson_cstr_bulk(string_t& /*result*/, std::false_type /*bulk*/) const noexcept + { + return false; + } + /*! @brief Parses a zero-terminated string of length @a len from the BSON input. @@ -15896,6 +15995,549 @@ class binary_reader } } + ////////// + // BON8 // + ////////// + + /*! + @brief get the next byte of a BON8 value + + A BON8 string has no length prefix and no mandatory terminator: it ends at + the first byte that cannot continue it, which is already the first byte (or, + for an integer that begins with a UTF-8 lead byte, the first two bytes) of + whatever follows. The string reader hands those bytes back with + @ref unget_bon8, and every BON8 read goes through this function so that + they are seen again. + + @return character read from the input + */ + char_int_type get_bon8() + { + if (bon8_pushback_size != 0) + { + ++chars_read; + return current = bon8_pushback[--bon8_pushback_size]; + } + return get(); + } + + /*! + @brief hand a byte back so that the next @ref get_bon8 returns it again + + @param[in] c the byte to hand back; bytes handed back are returned in + reverse order + */ + void unget_bon8(const char_int_type c) + { + // At most two bytes are ever handed back: a byte is only handed back + // right after it was read with get_bon8(), and the only place that + // hands back two bytes (a lead byte and the byte after it) read both + // of them in a row, which emptied the buffer first. This is an + // invariant of the reader rather than a property of the input, so + // an assertion suffices (the fuzzers are built with assertions). + JSON_ASSERT(bon8_pushback_size < bon8_pushback.size()); + bon8_pushback[bon8_pushback_size++] = c; + --chars_read; + } + + /*! + @param[in] c a byte + @return whether @a c is a UTF-8 continuation byte (0x80..0xBF) + */ + static constexpr bool is_bon8_continuation(const char_int_type c) noexcept + { + return 0x80 <= c && c <= 0xBF; + } + + /*! + @brief report a parse error at the last read byte + + @param[in] detail a detailed error message + @param[in] context further context information + @return false + */ + bool bon8_error(const std::string& detail, const char* context) + { + auto last_token = get_token_string(); + return sax->parse_error(chars_read, last_token, parse_error::create(112, chars_read, + exception_message(input_format_t::bon8, concat(detail, ": 0x", last_token), context), nullptr)); + } + + /*! + @brief read a BON8 value and everything nested inside it + + Reads values until the one that was begun here is complete, resuming the + enclosing container after each element, so that the nesting depth of the + input costs heap rather than native stack (see #5104). + + @return whether reading the value succeeded + */ + bool parse_bon8_internal() + { + // the key currently being read; hoisted out of the loop so that its + // capacity is reused across elements and across nesting levels + string_t key; + + while (true) + { + if (!container_stack.empty()) + { + // a copy, not a reference: it must stay valid across the + // pop_back() below, which destroys the container_stack element + // it would otherwise alias + const container_frame top = container_stack.back(); + bool at_end = false; + + if (top.remaining != npos) + { + // counted container (0x80..0x84, 0x86..0x8A): it ends once + // its elements have been read + at_end = (top.remaining == 0); + if (!at_end) + { + // claim the element about to be read + --container_stack.back().remaining; + } + } + else + { + // container 0x85 or 0x8B: it ends at an end-of-container + // marker (0xFE); any other byte begins the next element + at_end = (get_bon8() == 0xFE); + if (!at_end) + { + unget_bon8(current); + } + } + + if (at_end) + { + container_stack.pop_back(); + if (JSON_HEDLEY_UNLIKELY(top.is_object ? !sax->end_object() : !sax->end_array())) + { + return false; + } + // the value begun here is complete once its container is + if (container_stack.empty()) + { + return true; + } + continue; + } + + if (top.is_object) + { + key.clear(); + if (JSON_HEDLEY_UNLIKELY(!get_bon8_key(key) || !sax->key(key))) + { + return false; + } + } + } + + if (JSON_HEDLEY_UNLIKELY(!parse_bon8_value())) + { + return false; + } + + // a value that opened a container left it on the stack; one that + // did not, and that was not inside a container, was the whole value + if (container_stack.empty()) + { + return true; + } + } + } + + /*! + @brief read one BON8 value + + Reads a single value and passes it to the SAX parser. A value that begins + a container is not read to its end: the container is opened with + @ref enter_container and its elements are read by + @ref parse_bon8_internal, so that nesting does not consume native stack. + + @return whether reading the value succeeded + */ + bool parse_bon8_value() + { + const auto byte = get_bon8(); + + if (byte == char_traits::eof()) + { + return unexpect_eof(input_format_t::bon8, "value"); + } + + // string: ASCII character + if (byte <= 0x7F) + { + string_t s; + unget_bon8(byte); + return get_bon8_string(s) && sax->string(s); + } + + // array with 0..4 elements + if (byte <= 0x84) + { + return enter_array(static_cast(byte - 0x80)); + } + + // array terminated by 0xFE + if (byte == 0x85) + { + return enter_array(npos); + } + + // object with 0..4 members + if (byte <= 0x8A) + { + return enter_object(static_cast(byte - 0x86)); + } + + switch (byte) + { + case 0x8B: // object terminated by 0xFE + return enter_object(npos); + + case 0x8C: // int32 + { + std::int32_t number{}; + return get_number(input_format_t::bon8, number) && emit_bon8_integer(number); + } + + case 0x8D: // int64 + { + std::int64_t number{}; + return get_number(input_format_t::bon8, number) && emit_bon8_integer(number); + } + + case 0x8E: // binary32 + { + float number{}; + return get_number(input_format_t::bon8, number) && sax->number_float(static_cast(number), ""); + } + + case 0x8F: // binary64 + { + double number{}; + return get_number(input_format_t::bon8, number) && sax->number_float(static_cast(number), ""); + } + + case 0xF8: + return sax->boolean(false); + + case 0xF9: + return sax->boolean(true); + + case 0xFA: + return sax->null(); + + case 0xFB: + return sax->number_float(static_cast(-1.0), ""); + + case 0xFC: + return sax->number_float(static_cast(0.0), ""); + + case 0xFD: + return sax->number_float(static_cast(1.0), ""); + + case 0xFF: // empty string + { + string_t s; + return sax->string(s); + } + + default: + break; + } + + // integer 0..39 + if (byte <= 0xB7) + { + return sax->number_unsigned(static_cast(byte - 0x90)); + } + + // integer -1..-10 + if (byte <= 0xC1) + { + return sax->number_integer(-1 - static_cast(byte - 0xB8)); + } + + // 0xC2..0xF7: a UTF-8 lead byte begins a string if a continuation + // byte follows and an integer otherwise + if (byte <= 0xF7) + { + const auto second = get_bon8(); + if (is_bon8_continuation(second)) + { + string_t s; + unget_bon8(second); + unget_bon8(byte); + return get_bon8_string(s) && sax->string(s); + } + return get_bon8_integer(byte, second); + } + + // 0xFE: end of container where a value is expected + return bon8_error("invalid byte", "value"); + } + + /*! + @brief pass an integer to the SAX parser + + Non-negative integers are passed as unsigned, negative integers as signed + numbers, like the other binary formats do. + + @param[in] number the integer + @return whether the SAX parser accepted the value + */ + bool emit_bon8_integer(const std::int64_t number) + { + if (number >= 0) + { + return sax->number_unsigned(static_cast(number)); + } + return sax->number_integer(static_cast(number)); + } + + /*! + @brief read an integer encoded in 2..4 bytes + + The first byte is a UTF-8 lead byte (0xC2..0xF7) that is followed by a + byte that is not a continuation byte: 0x00..0x7F for positive and + 0xC0..0xFF for negative integers. The lead byte's low bits and the second + byte's low 7 (positive) or 6 (negative) bits are the most significant bits + of the value; 3- and 4-byte integers add one or two full bytes. Each range + starts where the shorter one ends, so no value has two encodings of the + same length. + + @param[in] lead the first byte (0xC2..0xF7) + @param[in] second the second byte + @return whether reading the integer succeeded + */ + bool get_bon8_integer(const char_int_type lead, const char_int_type second) + { + if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format_t::bon8, "number"))) + { + return false; + } + + const bool negative = second >= 0xC0; + auto value = static_cast(negative ? (second & 0x3F) : second); + std::int64_t offset = 0; + int extra_bytes = 0; + + if (lead <= 0xDF) + { + value |= static_cast(lead - 0xC2) << (negative ? 6 : 7); + offset = negative ? 11 : 40; + } + else if (lead <= 0xEF) + { + value |= static_cast(lead & 0x0F) << (negative ? 6 : 7); + offset = negative ? 1931 : 3880; + extra_bytes = 1; + } + else + { + value |= static_cast(lead & 0x07) << (negative ? 6 : 7); + offset = negative ? 264075 : 528168; + extra_bytes = 2; + } + + for (int i = 0; i < extra_bytes; ++i) + { + if (JSON_HEDLEY_UNLIKELY(get_bon8() == char_traits::eof())) + { + return unexpect_eof(input_format_t::bon8, "number"); + } + value = (value << 8) | static_cast(current); + } + + return negative ? sax->number_integer(static_cast(-(value + offset))) + : sax->number_unsigned(static_cast(value + offset)); + } + + /*! + @brief read an object key + + A key must be a string, so its first byte must be an ASCII character, a + UTF-8 lead byte followed by a continuation byte, or 0xFF (empty string). + + @param[out] result the key + @return whether reading the key succeeded + */ + bool get_bon8_key(string_t& result) + { + const auto byte = get_bon8(); + + if (byte == char_traits::eof()) + { + return unexpect_eof(input_format_t::bon8, "key"); + } + + if (byte == 0xFF) + { + return true; + } + + if (byte <= 0x7F) + { + unget_bon8(byte); + return get_bon8_string(result); + } + + if (0xC2 <= byte && byte <= 0xF7) + { + const auto second = get_bon8(); + unget_bon8(second); + if (is_bon8_continuation(second)) + { + unget_bon8(byte); + return get_bon8_string(result); + } + // an integer: report its first byte rather than the one after it + current = byte; + } + + return bon8_error("expected a string; last byte", "key"); + } + + /*! + @brief append the run of valid UTF-8 at the read position to a string + + For contiguous input, the ASCII characters and complete well-formed UTF-8 + sequences at the read position are appended to @a result in one step. The + byte that stops the run (an end-of-string marker, the first byte of the + next value, or an ill-formed byte) is left for @ref get_bon8_string, so + that strings end and errors are reported exactly as without this step. + + @param[in,out] result the string to append to + */ + void get_bon8_string_bulk(string_t& result, std::true_type /*bulk*/) + { + // bytes handed back must be read through get_bon8() first + if (bon8_pushback_size != 0) + { + return; + } + const std::size_t remaining = ia.bulk_remaining(); + if (remaining == 0) + { + return; + } + const auto* const data = reinterpret_cast(ia.bulk_data()); + const std::size_t length = valid_utf8_prefix(data, remaining); + if (length != 0) + { + result.append(reinterpret_cast(data), length); + ia.bulk_skip(length); + chars_read += length; + } + } + + /// input that is not contiguous: strings are read byte by byte + void get_bon8_string_bulk(string_t& /*result*/, std::false_type /*bulk*/) const noexcept {} + + /*! + @brief read a string + + Reads UTF-8 characters until an end-of-string marker (0xFF), which is + consumed, or a byte that cannot continue the string, which is handed back + to be read as the start of the next value. The string must be valid UTF-8, + and it must not end at the end of the input: the last string of a message + is always terminated by 0xFF. + + @param[out] result the string + @return whether reading the string succeeded + */ + bool get_bon8_string(string_t& result) + { + while (true) + { + get_bon8_string_bulk(result, std::integral_constant {}); + + const auto byte = get_bon8(); + + if (byte == char_traits::eof()) + { + return unexpect_eof(input_format_t::bon8, "string"); + } + + // end of string + if (byte == 0xFF) + { + return true; + } + + // ASCII character + if (byte <= 0x7F) + { + result.push_back(static_cast(byte)); + continue; + } + + // a byte that cannot begin a character ends the string and begins + // the next value + if (byte < 0xC2 || byte > 0xF7) + { + unget_bon8(byte); + return true; + } + + // a lead byte ends the string if no continuation byte follows: it + // is then the first byte of an integer + const auto second = get_bon8(); + if (!is_bon8_continuation(second)) + { + unget_bon8(second); + unget_bon8(byte); + return true; + } + + // the valid range of the second byte excludes overlong forms, + // surrogates, and code points above U+10FFFF + // (RFC 3629, section 4) + int continuation_bytes = 0; + bool valid_second = true; + if (byte <= 0xDF) + { + continuation_bytes = 1; + } + else if (byte <= 0xEF) + { + continuation_bytes = 2; + valid_second = (byte != 0xE0 || second >= 0xA0) && (byte != 0xED || second <= 0x9F); + } + else + { + continuation_bytes = 3; + valid_second = byte <= 0xF4 && (byte != 0xF0 || second >= 0x90) && (byte != 0xF4 || second <= 0x8F); + } + + if (JSON_HEDLEY_UNLIKELY(!valid_second)) + { + return bon8_error("invalid UTF-8 byte", "string"); + } + + result.push_back(static_cast(byte)); + result.push_back(static_cast(second)); + + for (int i = 1; i < continuation_bytes; ++i) + { + if (JSON_HEDLEY_UNLIKELY(get_bon8() == char_traits::eof())) + { + return unexpect_eof(input_format_t::bon8, "string"); + } + if (JSON_HEDLEY_UNLIKELY(!is_bon8_continuation(current))) + { + return bon8_error("invalid UTF-8 byte", "string"); + } + result.push_back(static_cast(current)); + } + } + } + /////////////////////// // Utility functions // /////////////////////// @@ -16194,6 +16836,10 @@ class binary_reader error_msg += "BJData"; break; + case input_format_t::bon8: + error_msg += "BON8"; + break; + case input_format_t::json: // LCOV_EXCL_LINE default: // LCOV_EXCL_LINE JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE @@ -16226,6 +16872,11 @@ class binary_reader /// the containers that have been opened and not closed yet; see @ref container_frame std::vector container_stack{}; + /// BON8: bytes read past the end of a string, returned again by @ref get_bon8 + std::array bon8_pushback{{}}; + /// BON8: number of bytes in @ref bon8_pushback + std::size_t bon8_pushback_size = 0; + // excluded markers in bjdata optimized type #define JSON_BINARY_READER_MAKE_BJD_OPTIMIZED_TYPE_MARKERS_ \ make_array('F', 'H', 'N', 'S', 'T', 'Z', '[', '{') @@ -19282,6 +19933,8 @@ NLOHMANN_JSON_NAMESPACE_END // #include +// #include + // #include // #include @@ -19556,7 +20209,7 @@ std::size_t binary_reserve_hint(const BasicJsonType& j) } /*! -@brief serialization to CBOR and MessagePack values +@brief serialization to BJData, BON8, BSON, CBOR, MessagePack, and UBJSON values */ template> class binary_writer @@ -20353,6 +21006,21 @@ class binary_writer } } + /*! + @param[in] j JSON value to serialize + */ + void write_bon8(const BasicJsonType& j) + { + bool string_open = false; + write_bon8_value(j, string_open); + + // the last string of a message must be terminated + if (string_open) + { + oa.write_character(to_char_type(0xFF)); + } + } + private: ////////// // BSON // @@ -20911,6 +21579,28 @@ class binary_writer return to_char_type(0xCB); // float 64 } + /// @return the BON8 type marker for binary32 (float) or binary64 (double) + template + static constexpr CharType get_bon8_float_prefix() + { + return to_char_type(std::is_same::value ? 0x8E : 0x8F); + } + + /// @return the type marker for a FloatType value in @a format (CBOR, MessagePack, or BON8) + template + static CharType get_compact_float_prefix(const detail::input_format_t format) + { + if (format == detail::input_format_t::cbor) + { + return get_cbor_float_prefix(FloatType{}); + } + if (format == detail::input_format_t::bon8) + { + return get_bon8_float_prefix(); + } + return get_msgpack_float_prefix(FloatType{}); + } + //////////// // UBJSON // //////////// @@ -21530,6 +22220,322 @@ class binary_writer return false; } + ////////// + // BON8 // + ////////// + + /*! + @brief write a BON8 value + + A string is written without length or terminator: it ends at the first + byte that cannot continue it, which is the first byte of any non-string + value and of the end-of-container marker 0xFE. It only needs an explicit + end-of-string marker (0xFF) when it is empty, when another string follows, + or when it is the last thing in the message. + + @param[in] j JSON value to serialize + @param[in,out] string_open whether the output ends with a non-empty + string that has not been terminated with 0xFF + */ + void write_bon8_value(const BasicJsonType& j, bool& string_open) + { + switch (j.type()) + { + case value_t::null: + { + write_bon8_marker(0xFA, string_open); + break; + } + + case value_t::boolean: + { + write_bon8_marker(j.m_data.m_value.boolean ? 0xF9 : 0xF8, string_open); + break; + } + + case value_t::number_unsigned: + { + if (j.m_data.m_value.number_unsigned > static_cast((std::numeric_limits::max)())) + { + JSON_THROW(out_of_range::create(407, concat("integer number ", std::to_string(j.m_data.m_value.number_unsigned), " cannot be represented by BON8 as it does not fit int64"), &j)); + } + write_bon8_integer(static_cast(j.m_data.m_value.number_unsigned)); + string_open = false; + break; + } + + case value_t::number_integer: + { + write_bon8_integer(static_cast(j.m_data.m_value.number_integer)); + string_open = false; + break; + } + + case value_t::number_float: + { + write_bon8_float(j.m_data.m_value.number_float); + string_open = false; + break; + } + + case value_t::string: + { + write_bon8_string(*j.m_data.m_value.string, string_open, j); + break; + } + + case value_t::array: + { + const auto N = j.m_data.m_value.array->size(); + // 0x80..0x84: array with 0..4 elements; 0x85: array ended by 0xFE + write_bon8_marker(static_cast(N <= 4 ? 0x80 + N : 0x85), string_open); + + for (const auto& el : *j.m_data.m_value.array) + { + write_bon8_value(el, string_open); + } + + if (N > 4) + { + write_bon8_marker(0xFE, string_open); + } + break; + } + + case value_t::object: + { + const auto N = j.m_data.m_value.object->size(); + // 0x86..0x8A: object with 0..4 members; 0x8B: object ended by 0xFE + write_bon8_marker(static_cast(N <= 4 ? 0x86 + N : 0x8B), string_open); + + for (const auto& el : *j.m_data.m_value.object) + { + write_bon8_string(el.first, string_open, j); + write_bon8_value(el.second, string_open); + } + + if (N > 4) + { + write_bon8_marker(0xFE, string_open); + } + break; + } + + case value_t::binary: + { + // BON8 has no binary type: write the bytes as an array of + // integers, like UBJSON and BJData do + const auto N = j.m_data.m_value.binary->size(); + write_bon8_marker(static_cast(N <= 4 ? 0x80 + N : 0x85), string_open); + + for (std::size_t i = 0; i < N; ++i) + { + // the cast is needed for binary types whose value type + // is not an integer (e.g., std::byte) + write_bon8_integer(static_cast(j.m_data.m_value.binary->data()[i])); + } + + if (N > 4) + { + write_bon8_marker(0xFE, string_open); + } + break; + } + + case value_t::discarded: + default: + break; + } + } + + /*! + @brief write a single byte that is not part of a string + + @param[in] marker the byte to write + @param[out] string_open set to false, because the output no longer ends + with a string; see @ref write_bon8_value + */ + void write_bon8_marker(const std::uint8_t marker, bool& string_open) + { + oa.write_character(to_char_type(marker)); + string_open = false; + } + + /*! + @brief write a string + + @param[in] s the string to write + @param[in,out] string_open see @ref write_bon8_value + @param[in] context the value the string belongs to (for diagnostics) + + @throw type_error.316 if @a s is not valid UTF-8, because the end of a + string is determined from its encoding + */ + void write_bon8_string(const string_t& s, bool& string_open, const BasicJsonType& context) + { + check_bon8_utf8(s, context); + + // a string that follows another string terminates it + if (string_open) + { + oa.write_character(to_char_type(0xFF)); + } + + if (s.empty()) + { + // the empty string is just the end-of-string marker + oa.write_character(to_char_type(0xFF)); + string_open = false; + } + else + { + oa.write_characters(reinterpret_cast(s.data()), s.size()); + string_open = true; + } + } + + /*! + @brief check that a string is valid UTF-8 (RFC 3629) + + @param[in] s the string to check + @param[in] context the value the string belongs to (for diagnostics) + + @throw type_error.316 if @a s is not valid UTF-8; the message names the + first byte of the first invalid or incomplete sequence + */ + static void check_bon8_utf8(const string_t& s, const BasicJsonType& context) + { + static_cast(context); // only used when exceptions are enabled + const auto* data = reinterpret_cast(s.data()); + const std::size_t valid = valid_utf8_prefix(data, s.size()); + if (JSON_HEDLEY_UNLIKELY(valid != s.size())) + { + JSON_THROW(type_error::create(316, concat("invalid UTF-8 byte at index ", std::to_string(valid), ": 0x", hex_byte(data[valid])), &context)); + } + } + + /// @return a byte as two uppercase hexadecimal digits + static std::string hex_byte(const std::uint8_t byte) + { + std::string result = "00"; + constexpr const char* nibble_to_hex = "0123456789ABCDEF"; + result[0] = nibble_to_hex[byte / 16]; + result[1] = nibble_to_hex[byte % 16]; + return result; + } + + /*! + @brief write an integer in the shortest encoding + + Integers from -10 to 39 take one byte. Up to -33818506 and 67637031, an + integer takes 2 to 4 bytes that begin with a UTF-8 lead byte (0xC2..0xF7) + followed by a byte that is not a continuation byte: 0x00..0x7F for + positive and 0xC0..0xFF for negative integers. Each range starts where the + shorter one ends. Larger integers are written as int32 (0x8C) or int64 + (0x8D) in big-endian byte order. + + @param[in] value the integer to write + */ + void write_bon8_integer(std::int64_t value) + { + if (value < (std::numeric_limits::min)() || value > (std::numeric_limits::max)()) + { + oa.write_character(to_char_type(0x8D)); + write_number(value); + } + else if (value < -33818506 || value > 67637031) + { + oa.write_character(to_char_type(0x8C)); + write_number(static_cast(value)); + } + else if (value <= -264075) + { + value = -(value + 264075); + write_bon8_bytes(0xF0 + ((value >> 22) & 0x07), 0xC0 + ((value >> 16) & 0x3F), value >> 8, value); + } + else if (value <= -1931) + { + value = -(value + 1931); + write_bon8_bytes(0xE0 + ((value >> 14) & 0x0F), 0xC0 + ((value >> 8) & 0x3F), value); + } + else if (value <= -11) + { + value = -(value + 11); + write_bon8_bytes(0xC2 + ((value >> 6) & 0x1F), 0xC0 + (value & 0x3F)); + } + else if (value <= -1) + { + write_bon8_bytes(0xB8 - (value + 1)); + } + else if (value <= 39) + { + write_bon8_bytes(0x90 + value); + } + else if (value <= 3879) + { + value -= 40; + write_bon8_bytes(0xC2 + ((value >> 7) & 0x1F), value & 0x7F); + } + else if (value <= 528167) + { + value -= 3880; + write_bon8_bytes(0xE0 + ((value >> 15) & 0x0F), (value >> 8) & 0x7F, value); + } + else + { + value -= 528168; + write_bon8_bytes(0xF0 + ((value >> 23) & 0x07), (value >> 16) & 0x7F, value >> 8, value); + } + } + + /// write the low byte of each argument + template + void write_bon8_bytes(const Bytes... bytes) + { + const std::array buffer{{to_char_type(static_cast(bytes & 0xFF))...}}; + oa.write_characters(buffer.data(), buffer.size()); + } + + /*! + @brief write a floating-point number + + -1.0, +0.0, and 1.0 take one byte. Other numbers are written as binary32 + (0x8E) if that loses no precision, and as binary64 (0x8F) otherwise; -0.0, + infinities, and NaN are always written as binary32, NaN as 0x7F800001. + + @param[in] n the number to write + */ + void write_bon8_float(const number_float_t n) + { +#ifdef __GNUC__ + JSON_HEDLEY_DIAGNOSTIC_PUSH + JSON_HEDLEY_PRAGMA(GCC diagnostic ignored "-Wfloat-equal") +#endif + if (n == static_cast(-1)) + { + oa.write_character(to_char_type(0xFB)); + } + else if (n == static_cast(0) && !std::signbit(n)) + { + oa.write_character(to_char_type(0xFC)); + } + else if (n == static_cast(1)) + { + oa.write_character(to_char_type(0xFD)); + } + else if (std::isnan(n)) + { + write_bon8_bytes(0x8E, 0x7F, 0x80, 0x00, 0x01); + } + else + { + write_compact_float(n, detail::input_format_t::bon8); + } +#ifdef __GNUC__ + JSON_HEDLEY_DIAGNOSTIC_POP +#endif + } + /////////////////////// // Utility functions // /////////////////////// @@ -21664,16 +22670,12 @@ class binary_writer static_cast(n) <= static_cast((std::numeric_limits::max)()) && static_cast(static_cast(n)) == static_cast(n)))) { - oa.write_character(format == detail::input_format_t::cbor - ? get_cbor_float_prefix(static_cast(n)) - : get_msgpack_float_prefix(static_cast(n))); + oa.write_character(get_compact_float_prefix(format)); write_number(static_cast(n)); } else { - oa.write_character(format == detail::input_format_t::cbor - ? get_cbor_float_prefix(n) - : get_msgpack_float_prefix(n)); + oa.write_character(get_compact_float_prefix(format)); write_number(n); } #ifdef __GNUC__ @@ -30171,6 +31173,30 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec binary_writer(o).write_bson(j); } + /// @brief create a BON8 serialization of a given JSON value + /// @sa https://json.nlohmann.me/api/basic_json/to_bon8/ + static std::vector to_bon8(const basic_json& j) + { + std::vector result; + result.reserve(detail::binary_reserve_hint(j)); + vector_writer(result).write_bon8(j); + return result; + } + + /// @brief create a BON8 serialization of a given JSON value + /// @sa https://json.nlohmann.me/api/basic_json/to_bon8/ + static void to_bon8(const basic_json& j, detail::output_adapter o) + { + binary_writer(o).write_bon8(j); + } + + /// @brief create a BON8 serialization of a given JSON value + /// @sa https://json.nlohmann.me/api/basic_json/to_bon8/ + static void to_bon8(const basic_json& j, detail::output_adapter o) + { + binary_writer(o).write_bon8(j); + } + /// @brief create a JSON value from an input in CBOR format /// @sa https://json.nlohmann.me/api/basic_json/from_cbor/ template @@ -30404,6 +31430,43 @@ class basic_json // NOLINT(cppcoreguidelines-special-member-functions,hicpp-spec return result; } + /// @brief create a JSON value from an input in BON8 format + /// @sa https://json.nlohmann.me/api/basic_json/from_bon8/ + template + JSON_HEDLEY_WARN_UNUSED_RESULT + static basic_json from_bon8(InputType&& i, + const bool strict = true, + const bool allow_exceptions = true) + { + basic_json result; + auto ia = detail::input_adapter(std::forward(i)); + detail::json_sax_dom_parser sdp(result, allow_exceptions); + if (!binary_reader(std::move(ia), input_format_t::bon8).sax_parse(input_format_t::bon8, &sdp, strict)) // cppcheck-suppress[accessMoved] + { + result = value_t::discarded; + } + return result; + } + + /// @brief create a JSON value from an input in BON8 format (iterator pair, or iterator+sentinel pair for C++20 ranges support) + /// @sa https://json.nlohmann.me/api/basic_json/from_bon8/ + template::value, int> = 0> + JSON_HEDLEY_WARN_UNUSED_RESULT + static basic_json from_bon8(IteratorType first, SentinelType last, + const bool strict = true, + const bool allow_exceptions = true) + { + basic_json result; + auto ia = detail::input_adapter(std::move(first), std::move(last)); + detail::json_sax_dom_parser sdp(result, allow_exceptions); + if (!binary_reader(std::move(ia), input_format_t::bon8).sax_parse(input_format_t::bon8, &sdp, strict)) // cppcheck-suppress[accessMoved] + { + result = value_t::discarded; + } + return result; + } + /// @brief create a JSON value from an input in BSON format /// @sa https://json.nlohmann.me/api/basic_json/from_bson/ template diff --git a/tests/CMakeLists.txt b/tests/CMakeLists.txt index ea6da107c..8c81b9123 100644 --- a/tests/CMakeLists.txt +++ b/tests/CMakeLists.txt @@ -112,7 +112,7 @@ endif() if (CMAKE_CXX_COMPILER_ID STREQUAL "MSVC") # avoid stack overflow, see https://github.com/nlohmann/json/issues/2955 - json_test_set_test_options("test-cbor;test-msgpack;test-ubjson;test-bjdata;test-binary_formats" LINK_OPTIONS /STACK:4000000) + json_test_set_test_options("test-bon8;test-cbor;test-msgpack;test-ubjson;test-bjdata;test-binary_formats" LINK_OPTIONS /STACK:4000000) endif() # disable exceptions for test-disabled_exceptions diff --git a/tests/Makefile b/tests/Makefile index 3a11ce7dd..6bfa14397 100644 --- a/tests/Makefile +++ b/tests/Makefile @@ -10,7 +10,7 @@ CXXFLAGS += -std=c++11 CPPFLAGS += -I ../single_include FUZZER_ENGINE = src/fuzzer-driver_afl.cpp -FUZZERS = parse_afl_fuzzer parse_bson_fuzzer parse_cbor_fuzzer parse_msgpack_fuzzer parse_ubjson_fuzzer parse_bjdata_fuzzer +FUZZERS = parse_afl_fuzzer parse_bson_fuzzer parse_cbor_fuzzer parse_msgpack_fuzzer parse_ubjson_fuzzer parse_bjdata_fuzzer parse_bon8_fuzzer fuzzers: $(FUZZERS) parse_afl_fuzzer: @@ -30,3 +30,6 @@ parse_ubjson_fuzzer: parse_bjdata_fuzzer: $(CXX) $(CXXFLAGS) $(CPPFLAGS) $(FUZZER_ENGINE) src/fuzzer-parse_bjdata.cpp -o $@ + +parse_bon8_fuzzer: + $(CXX) $(CXXFLAGS) $(CPPFLAGS) $(FUZZER_ENGINE) src/fuzzer-parse_bon8.cpp -o $@ diff --git a/tests/benchmarks/src/benchmarks.cpp b/tests/benchmarks/src/benchmarks.cpp index 613ca1baa..2ad28a57a 100644 --- a/tests/benchmarks/src/benchmarks.cpp +++ b/tests/benchmarks/src/benchmarks.cpp @@ -274,7 +274,8 @@ enum class binary_format ubjson_optimized, bjdata, bjdata_optimized, - bson + bson, + bon8 }; static std::vector to_binary(const json& j, const binary_format format) @@ -293,6 +294,8 @@ static std::vector to_binary(const json& j, const binary_format fo return json::to_bjdata(j); case binary_format::bjdata_optimized: return json::to_bjdata(j, true, true); + case binary_format::bon8: + return json::to_bon8(j); case binary_format::bson: default: return json::to_bson(j); @@ -313,6 +316,8 @@ static json from_binary(const std::vector& bytes, const binary_for case binary_format::bjdata: case binary_format::bjdata_optimized: return json::from_bjdata(bytes); + case binary_format::bon8: + return json::from_bon8(bytes); case binary_format::bson: default: return json::from_bson(bytes); @@ -333,6 +338,8 @@ static json from_binary(std::FILE* file, const binary_format format) case binary_format::bjdata: case binary_format::bjdata_optimized: return json::from_bjdata(file); + case binary_format::bon8: + return json::from_bon8(file); case binary_format::bson: default: return json::from_bson(file); @@ -407,6 +414,10 @@ BENCHMARK_CAPTURE(FromBinaryBuffer, bjdata / canada, TEST_DATA_DIRECTORY "/nativ BENCHMARK_CAPTURE(FromBinaryBuffer, bjdata / twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", binary_format::bjdata); BENCHMARK_CAPTURE(FromBinaryBuffer, bjdata_optimized / canada, TEST_DATA_DIRECTORY "/nativejson-benchmark/canada.json", binary_format::bjdata_optimized); BENCHMARK_CAPTURE(FromBinaryBuffer, bjdata_optimized / twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", binary_format::bjdata_optimized); +BENCHMARK_CAPTURE(FromBinaryBuffer, bon8 / jeopardy, TEST_DATA_DIRECTORY "/jeopardy/jeopardy.json", binary_format::bon8); +BENCHMARK_CAPTURE(FromBinaryBuffer, bon8 / canada, TEST_DATA_DIRECTORY "/nativejson-benchmark/canada.json", binary_format::bon8); +BENCHMARK_CAPTURE(FromBinaryBuffer, bon8 / citm_catalog, TEST_DATA_DIRECTORY "/nativejson-benchmark/citm_catalog.json", binary_format::bon8); +BENCHMARK_CAPTURE(FromBinaryBuffer, bon8 / twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", binary_format::bon8); // BSON requires an object at the top level, so the array-rooted test files // (jeopardy and the regression files) cannot be captured here BENCHMARK_CAPTURE(FromBinaryBuffer, bson / canada, TEST_DATA_DIRECTORY "/nativejson-benchmark/canada.json", binary_format::bson); @@ -450,6 +461,8 @@ BENCHMARK_CAPTURE(FromBinaryFile, cbor / twitter, TEST_DATA_DIRECTORY "/nativejs BENCHMARK_CAPTURE(FromBinaryFile, ubjson / canada, TEST_DATA_DIRECTORY "/nativejson-benchmark/canada.json", binary_format::ubjson); BENCHMARK_CAPTURE(FromBinaryFile, ubjson / twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", binary_format::ubjson); BENCHMARK_CAPTURE(FromBinaryFile, bjdata / twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", binary_format::bjdata); +BENCHMARK_CAPTURE(FromBinaryFile, bon8 / canada, TEST_DATA_DIRECTORY "/nativejson-benchmark/canada.json", binary_format::bon8); +BENCHMARK_CAPTURE(FromBinaryFile, bon8 / twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", binary_format::bon8); BENCHMARK_CAPTURE(FromBinaryFile, bson / twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", binary_format::bson); ////////////////////////////////////////////////////////////////////////////// @@ -530,18 +543,21 @@ BENCHMARK_CAPTURE(FromBinaryShape, nested / msgpack, make_nested, binary_format: BENCHMARK_CAPTURE(FromBinaryShape, nested / ubjson, make_nested, binary_format::ubjson); BENCHMARK_CAPTURE(FromBinaryShape, nested / bjdata, make_nested, binary_format::bjdata); BENCHMARK_CAPTURE(FromBinaryShape, nested / bson, make_nested, binary_format::bson); +BENCHMARK_CAPTURE(FromBinaryShape, nested / bon8, make_nested, binary_format::bon8); BENCHMARK_CAPTURE(FromBinaryShape, containers / cbor, make_containers, binary_format::cbor); BENCHMARK_CAPTURE(FromBinaryShape, containers / msgpack, make_containers, binary_format::msgpack); BENCHMARK_CAPTURE(FromBinaryShape, containers / ubjson, make_containers, binary_format::ubjson); BENCHMARK_CAPTURE(FromBinaryShape, containers / ubjson_optimized, make_containers, binary_format::ubjson_optimized); BENCHMARK_CAPTURE(FromBinaryShape, containers / bjdata, make_containers, binary_format::bjdata); BENCHMARK_CAPTURE(FromBinaryShape, containers / bson, make_containers, binary_format::bson); +BENCHMARK_CAPTURE(FromBinaryShape, containers / bon8, make_containers, binary_format::bon8); // BSON names every array element, so a large array measures key generation // rather than scalar decoding and is left out here BENCHMARK_CAPTURE(FromBinaryShape, scalars / cbor, make_scalars, binary_format::cbor); BENCHMARK_CAPTURE(FromBinaryShape, scalars / msgpack, make_scalars, binary_format::msgpack); BENCHMARK_CAPTURE(FromBinaryShape, scalars / ubjson, make_scalars, binary_format::ubjson); BENCHMARK_CAPTURE(FromBinaryShape, scalars / bjdata, make_scalars, binary_format::bjdata); +BENCHMARK_CAPTURE(FromBinaryShape, scalars / bon8, make_scalars, binary_format::bon8); /*! @brief parse an indefinite-length CBOR string diff --git a/tests/fuzzing.md b/tests/fuzzing.md index b3bf90d5d..49ab7b575 100644 --- a/tests/fuzzing.md +++ b/tests/fuzzing.md @@ -1,6 +1,6 @@ # Fuzz testing -Each parser of the library (JSON, BJData, BSON, CBOR, MessagePack, and UBJSON) can be fuzz tested. Currently, +Each parser of the library (JSON, BJData, BON8, BSON, CBOR, MessagePack, and UBJSON) can be fuzz tested. Currently, [libFuzzer](https://llvm.org/docs/LibFuzzer.html) and [afl++](https://github.com/AFLplusplus/AFLplusplus) are supported. ## Corpus creation @@ -10,11 +10,11 @@ directory with some simple input files that cover several features of the parser for mutations. ```shell -TEST_DATA_VERSION=3.1.0 +TEST_DATA_VERSION=3.2.0 wget https://github.com/nlohmann/json_test_data/archive/refs/tags/v$TEST_DATA_VERSION.zip unzip v$TEST_DATA_VERSION.zip rm v$TEST_DATA_VERSION.zip -for FORMAT in json bjdata bson cbor msgpack ubjson +for FORMAT in json bjdata bon8 bson cbor msgpack ubjson do rm -fr corpus_$FORMAT mkdir corpus_$FORMAT diff --git a/tests/src/fuzzer-parse_bon8.cpp b/tests/src/fuzzer-parse_bon8.cpp new file mode 100644 index 000000000..e97d4f17b --- /dev/null +++ b/tests/src/fuzzer-parse_bon8.cpp @@ -0,0 +1,103 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +/* +This file implements a parser test suitable for fuzz testing. Given a byte +array data, it performs the following steps: + +- j1 = from_bon8(data) +- vec = to_bon8(j1) +- j2 = from_bon8(vec) +- assert(j1 == j2) + +It also checks that reading the data from a stream, which reads strings byte by +byte, gives the same value or error as reading it from contiguous memory, which +copies strings in bulk. + +The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer +drivers. +*/ + +#include +#include +#include +#include + +// the round-trip checks below are assertions; NDEBUG would compile them away +#ifdef NDEBUG + #error "the fuzzer drivers must be built without NDEBUG" +#endif + +using json = nlohmann::json; + +namespace +{ +// the serialization of the value read from @a input, or the error message +template +std::string read_bon8(InputType&& input) +{ + try + { + const auto vec = json::to_bon8(json::from_bon8(std::forward(input))); + return {vec.begin(), vec.end()}; + } + catch (const json::exception& e) + { + return e.what(); + } +} +} // namespace + +// see http://llvm.org/docs/LibFuzzer.html +extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size) +{ + // contiguous and stream input must be read alike + { + std::istringstream stream(std::string(reinterpret_cast(data), size)); + assert(read_bon8(std::vector(data, data + size)) == read_bon8(stream)); + } + + try + { + // step 1: parse input + std::vector const vec1(data, data + size); + json const j1 = json::from_bon8(vec1); + + try + { + // step 2: round trip + std::vector const vec2 = json::to_bon8(j1); + + // parse serialization + json const j2 = json::from_bon8(vec2); + + // serializations must match + assert(json::to_bon8(j2) == vec2); + } + catch (const json::parse_error&) + { + // parsing a BON8 serialization must not fail + assert(false); + } + } + catch (const json::parse_error&) + { + // parse errors are ok, because input may be random bytes + } + catch (const json::type_error&) + { + // type errors can occur during parsing, too + } + catch (const json::out_of_range&) + { + // out of range errors may happen if provided sizes are excessive + } + + // return 0 - non-zero return values are reserved for future use + return 0; +} diff --git a/tests/src/unit-alt-string.cpp b/tests/src/unit-alt-string.cpp index ec2ac146b..cddaadb7e 100644 --- a/tests/src/unit-alt-string.cpp +++ b/tests/src/unit-alt-string.cpp @@ -185,6 +185,7 @@ TEST_CASE("alternative string type") CHECK(alt_json::from_cbor(alt_json::to_cbor(doc)) == doc); CHECK(alt_json::from_msgpack(alt_json::to_msgpack(doc)) == doc); + CHECK(alt_json::from_bon8(alt_json::to_bon8(doc)) == doc); // BSON is not covered: it additionally needs string_t::find(value_type), // which alt_string does not provide CHECK(alt_json::from_ubjson(alt_json::to_ubjson(doc)) == doc); diff --git a/tests/src/unit-binary_formats.cpp b/tests/src/unit-binary_formats.cpp index 0c084b720..ed6d89911 100644 --- a/tests/src/unit-binary_formats.cpp +++ b/tests/src/unit-binary_formats.cpp @@ -25,6 +25,7 @@ TEST_CASE("Binary Formats" * doctest::skip()) const auto bjdata_1_size = json::to_bjdata(j).size(); const auto bjdata_2_size = json::to_bjdata(j, true).size(); const auto bjdata_3_size = json::to_bjdata(j, true, true).size(); + const auto bon8_size = json::to_bon8(j).size(); const auto bson_size = json::to_bson(j).size(); const auto cbor_size = json::to_cbor(j).size(); const auto msgpack_size = json::to_msgpack(j).size(); @@ -36,6 +37,7 @@ TEST_CASE("Binary Formats" * doctest::skip()) CHECK(bjdata_1_size == 1112030); CHECK(bjdata_2_size == 1224148); CHECK(bjdata_3_size == 1224148); + CHECK(bon8_size == 1055792); CHECK(bson_size == 1794522); CHECK(cbor_size == 1055552); CHECK(msgpack_size == 1056145); @@ -47,6 +49,7 @@ TEST_CASE("Binary Formats" * doctest::skip()) CHECK((100.0 * double(bjdata_1_size) / double(json_size)) == Approx(53.199)); CHECK((100.0 * double(bjdata_2_size) / double(json_size)) == Approx(58.563)); CHECK((100.0 * double(bjdata_3_size) / double(json_size)) == Approx(58.563)); + CHECK((100.0 * double(bon8_size) / double(json_size)) == Approx(50.509)); CHECK((100.0 * double(bson_size) / double(json_size)) == Approx(85.849)); CHECK((100.0 * double(cbor_size) / double(json_size)) == Approx(50.497)); CHECK((100.0 * double(msgpack_size) / double(json_size)) == Approx(50.526)); @@ -64,6 +67,7 @@ TEST_CASE("Binary Formats" * doctest::skip()) const auto bjdata_1_size = json::to_bjdata(j).size(); const auto bjdata_2_size = json::to_bjdata(j, true).size(); const auto bjdata_3_size = json::to_bjdata(j, true, true).size(); + const auto bon8_size = json::to_bon8(j).size(); const auto bson_size = json::to_bson(j).size(); const auto cbor_size = json::to_cbor(j).size(); const auto msgpack_size = json::to_msgpack(j).size(); @@ -75,6 +79,7 @@ TEST_CASE("Binary Formats" * doctest::skip()) CHECK(bjdata_1_size == 425342); CHECK(bjdata_2_size == 429970); CHECK(bjdata_3_size == 429970); + CHECK(bon8_size == 391396); CHECK(bson_size == 444568); CHECK(cbor_size == 402814); CHECK(msgpack_size == 401510); @@ -86,6 +91,7 @@ TEST_CASE("Binary Formats" * doctest::skip()) CHECK((100.0 * double(bjdata_1_size) / double(json_size)) == Approx(91.097)); CHECK((100.0 * double(bjdata_2_size) / double(json_size)) == Approx(92.089)); CHECK((100.0 * double(bjdata_3_size) / double(json_size)) == Approx(92.089)); + CHECK((100.0 * double(bon8_size) / double(json_size)) == Approx(83.828)); CHECK((100.0 * double(bson_size) / double(json_size)) == Approx(95.215)); CHECK((100.0 * double(cbor_size) / double(json_size)) == Approx(86.273)); CHECK((100.0 * double(msgpack_size) / double(json_size)) == Approx(85.993)); @@ -103,6 +109,7 @@ TEST_CASE("Binary Formats" * doctest::skip()) const auto bjdata_1_size = json::to_bjdata(j).size(); const auto bjdata_2_size = json::to_bjdata(j, true).size(); const auto bjdata_3_size = json::to_bjdata(j, true, true).size(); + const auto bon8_size = json::to_bon8(j).size(); const auto bson_size = json::to_bson(j).size(); const auto cbor_size = json::to_cbor(j).size(); const auto msgpack_size = json::to_msgpack(j).size(); @@ -114,6 +121,7 @@ TEST_CASE("Binary Formats" * doctest::skip()) CHECK(bjdata_1_size == 390781); CHECK(bjdata_2_size == 433557); CHECK(bjdata_3_size == 432964); + CHECK(bon8_size == 317879); CHECK(bson_size == 479430); CHECK(cbor_size == 342373); CHECK(msgpack_size == 342473); @@ -125,6 +133,7 @@ TEST_CASE("Binary Formats" * doctest::skip()) CHECK((100.0 * double(bjdata_1_size) / double(json_size)) == Approx(78.109)); CHECK((100.0 * double(bjdata_2_size) / double(json_size)) == Approx(86.659)); CHECK((100.0 * double(bjdata_3_size) / double(json_size)) == Approx(86.541)); + CHECK((100.0 * double(bon8_size) / double(json_size)) == Approx(63.538)); CHECK((100.0 * double(bson_size) / double(json_size)) == Approx(95.828)); CHECK((100.0 * double(cbor_size) / double(json_size)) == Approx(68.433)); CHECK((100.0 * double(msgpack_size) / double(json_size)) == Approx(68.453)); @@ -142,6 +151,7 @@ TEST_CASE("Binary Formats" * doctest::skip()) const auto bjdata_1_size = json::to_bjdata(j).size(); const auto bjdata_2_size = json::to_bjdata(j, true).size(); const auto bjdata_3_size = json::to_bjdata(j, true, true).size(); + const auto bon8_size = json::to_bon8(j).size(); const auto bson_size = json::to_bson({{"", j}}).size(); // wrap array in object for BSON const auto cbor_size = json::to_cbor(j).size(); const auto msgpack_size = json::to_msgpack(j).size(); @@ -153,6 +163,7 @@ TEST_CASE("Binary Formats" * doctest::skip()) CHECK(bjdata_1_size == 50710965); CHECK(bjdata_2_size == 51144830); CHECK(bjdata_3_size == 51144830); + CHECK(bon8_size == 45942080); CHECK(bson_size == 56008520); CHECK(cbor_size == 46187320); CHECK(msgpack_size == 46158575); @@ -164,6 +175,7 @@ TEST_CASE("Binary Formats" * doctest::skip()) CHECK((100.0 * double(bjdata_1_size) / double(json_size)) == Approx(96.576)); CHECK((100.0 * double(bjdata_2_size) / double(json_size)) == Approx(97.402)); CHECK((100.0 * double(bjdata_3_size) / double(json_size)) == Approx(97.402)); + CHECK((100.0 * double(bon8_size) / double(json_size)) == Approx(87.494)); CHECK((100.0 * double(bson_size) / double(json_size)) == Approx(106.665)); CHECK((100.0 * double(cbor_size) / double(json_size)) == Approx(87.961)); CHECK((100.0 * double(msgpack_size) / double(json_size)) == Approx(87.906)); @@ -181,6 +193,7 @@ TEST_CASE("Binary Formats" * doctest::skip()) const auto bjdata_1_size = json::to_bjdata(j).size(); const auto bjdata_2_size = json::to_bjdata(j, true).size(); const auto bjdata_3_size = json::to_bjdata(j, true, true).size(); + const auto bon8_size = json::to_bon8(j).size(); // BSON cannot process the file as it contains code point U+0000 const auto cbor_size = json::to_cbor(j).size(); const auto msgpack_size = json::to_msgpack(j).size(); @@ -192,6 +205,7 @@ TEST_CASE("Binary Formats" * doctest::skip()) CHECK(bjdata_1_size == 148695); CHECK(bjdata_2_size == 150569); CHECK(bjdata_3_size == 150569); + CHECK(bon8_size == 144477); CHECK(cbor_size == 147095); CHECK(msgpack_size == 147017); CHECK(ubjson_1_size == 148695); @@ -202,6 +216,7 @@ TEST_CASE("Binary Formats" * doctest::skip()) CHECK((100.0 * double(bjdata_1_size) / double(json_size)) == Approx(88.153)); CHECK((100.0 * double(bjdata_2_size) / double(json_size)) == Approx(89.264)); CHECK((100.0 * double(bjdata_3_size) / double(json_size)) == Approx(89.264)); + CHECK((100.0 * double(bon8_size) / double(json_size)) == Approx(85.653)); CHECK((100.0 * double(cbor_size) / double(json_size)) == Approx(87.205)); CHECK((100.0 * double(msgpack_size) / double(json_size)) == Approx(87.158)); CHECK((100.0 * double(ubjson_1_size) / double(json_size)) == Approx(88.153)); diff --git a/tests/src/unit-binary_writer_sinks.cpp b/tests/src/unit-binary_writer_sinks.cpp index f60e1bf51..d13f5e5ac 100644 --- a/tests/src/unit-binary_writer_sinks.cpp +++ b/tests/src/unit-binary_writer_sinks.cpp @@ -12,6 +12,7 @@ using nlohmann::json; #include +#include #include #include @@ -49,6 +50,12 @@ std::vector test_values() }; } +// BON8 has no integers above the int64 range, so to_bon8() rejects them +bool bon8_representable(const json& j) +{ + return !j.is_number_unsigned() || j.get() <= static_cast((std::numeric_limits::max)()); +} + // values to_bson() accepts: the document must be an object std::vector bson_values() { @@ -92,6 +99,13 @@ TEST_CASE("binary writer output sinks") json::to_msgpack(j, msgpack); CHECK(json::to_msgpack(j) == msgpack); + if (bon8_representable(j)) + { + std::vector bon8; + json::to_bon8(j, bon8); + CHECK(json::to_bon8(j) == bon8); + } + for (const bool use_size : { false, true @@ -172,6 +186,10 @@ TEST_CASE("binary_reserve_hint never over-reserves") CHECK(hint <= json::to_ubjson(j).size()); CHECK(hint <= json::to_ubjson(j, true, true).size()); CHECK(hint <= json::to_bjdata(j).size()); + if (bon8_representable(j)) + { + CHECK(hint <= json::to_bon8(j).size()); + } } for (const auto& j : bson_values()) diff --git a/tests/src/unit-bon8.cpp b/tests/src/unit-bon8.cpp new file mode 100644 index 000000000..030e1d5c3 --- /dev/null +++ b/tests/src/unit-bon8.cpp @@ -0,0 +1,1050 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#include "doctest_compatibility.h" + +#include +using nlohmann::json; +#ifdef JSON_TEST_NO_GLOBAL_UDLS + using namespace nlohmann::literals; // NOLINT(google-build-using-namespace) +#endif + +#include +#include +#include +#include +#include +#include +#include "make_test_data_available.hpp" +#include "test_utils.hpp" + +namespace +{ +class SaxCountdown +{ + public: + explicit SaxCountdown(const int count) : events_left(count) + {} + + bool null() + { + return events_left-- > 0; + } + + bool boolean(bool /*unused*/) + { + return events_left-- > 0; + } + + bool number_integer(json::number_integer_t /*unused*/) + { + return events_left-- > 0; + } + + bool number_unsigned(json::number_unsigned_t /*unused*/) + { + return events_left-- > 0; + } + + bool number_float(json::number_float_t /*unused*/, const std::string& /*unused*/) + { + return events_left-- > 0; + } + + bool string(std::string& /*unused*/) + { + return events_left-- > 0; + } + + bool binary(std::vector& /*unused*/) + { + return events_left-- > 0; + } + + bool start_object(std::size_t /*unused*/) + { + return events_left-- > 0; + } + + bool key(std::string& /*unused*/) + { + return events_left-- > 0; + } + + bool end_object() + { + return events_left-- > 0; + } + + bool start_array(std::size_t /*unused*/) + { + return events_left-- > 0; + } + + bool end_array() + { + return events_left-- > 0; + } + + bool parse_error(std::size_t /*unused*/, const std::string& /*unused*/, const json::exception& /*unused*/) // NOLINT(readability-convert-member-functions-to-static) + { + return false; + } + + private: + int events_left = 0; +}; + +using bytes = std::vector; + +/// @return the string with the given bytes +std::string str(const bytes& b) +{ + // converted one by one: constructing the string from the byte range + // converts implicitly, which -fsanitize=integer reports for bytes >= 0x80 + std::string result; + result.reserve(b.size()); + for (const auto c : b) + { + result.push_back(static_cast(c)); + } + return result; +} + +/// check that @a j is serialized to @a expected and that @a expected is read back as @a j +void check_bon8(const json& j, const bytes& expected) +{ + CAPTURE(j) + CHECK(json::to_bon8(j) == expected); + + const json decoded = json::from_bon8(expected); + CHECK(decoded == j); + // integers are not read back as floats and vice versa + CHECK(decoded.type() == j.type()); + + // a stream is read byte by byte rather than in bulk + std::istringstream stream(str(expected)); + CHECK(json::from_bon8(stream) == decoded); +} + +/// @return @a b followed by @a tail +bytes concat(bytes b, const bytes& tail) +{ + b.insert(b.end(), tail.begin(), tail.end()); + return b; +} +} // namespace + +TEST_CASE("BON8") +{ + SECTION("individual values") + { + SECTION("discarded") + { + // discarded values are not serialized + const json j = json::value_t::discarded; + CHECK(json::to_bon8(j).empty()); + } + + SECTION("null") + { + check_bon8(nullptr, {0xFA}); + } + + SECTION("boolean") + { + check_bon8(true, {0xF9}); + check_bon8(false, {0xF8}); + } + + SECTION("integers") + { + // test vectors from HikoGUI (src/hikogui/codec/BON8_tests.cpp, + // Copyright Take Vos 2021, Boost Software License 1.0), which + // cover the first and last value of each encoding + SECTION("positive") + { + check_bon8(0u, {0x90}); + check_bon8(39u, {0xB7}); + check_bon8(40u, {0xC2, 0x00}); + check_bon8(3879u, {0xDF, 0x7F}); + check_bon8(3880u, {0xE0, 0x00, 0x00}); + check_bon8(528167u, {0xEF, 0x7F, 0xFF}); + check_bon8(528168u, {0xF0, 0x00, 0x00, 0x00}); + check_bon8(67637031u, {0xF7, 0x7F, 0xFF, 0xFF}); + check_bon8(67637032u, {0x8C, 0x04, 0x08, 0x0F, 0x28}); + check_bon8(2147483647u, {0x8C, 0x7F, 0xFF, 0xFF, 0xFF}); + check_bon8(2147483648u, {0x8D, 0x00, 0x00, 0x00, 0x00, 0x80, 0x00, 0x00, 0x00}); + check_bon8(9223372036854775807u, {0x8D, 0x7F, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF}); + } + + SECTION("negative") + { + check_bon8(-1, {0xB8}); + check_bon8(-10, {0xC1}); + check_bon8(-11, {0xC2, 0xC0}); + check_bon8(-1930, {0xDF, 0xFF}); + check_bon8(-1931, {0xE0, 0xC0, 0x00}); + check_bon8(-264074, {0xEF, 0xFF, 0xFF}); + check_bon8(-264075, {0xF0, 0xC0, 0x00, 0x00}); + check_bon8(-33818506, {0xF7, 0xFF, 0xFF, 0xFF}); + check_bon8(-33818507, {0x8C, 0xFD, 0xFB, 0xF8, 0x75}); + check_bon8(-2147483648LL, {0x8C, 0x80, 0x00, 0x00, 0x00}); + check_bon8(-2147483649LL, {0x8D, 0xFF, 0xFF, 0xFF, 0xFF, 0x7F, 0xFF, 0xFF, 0xFF}); + check_bon8((std::numeric_limits::min)(), {0x8D, 0x80, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00}); + } + + SECTION("values inside the ranges") + { + check_bon8(1u, {0x91}); + check_bon8(100u, {0xC2, 0x3C}); + check_bon8(1000u, {0xC9, 0x40}); + check_bon8(100000u, {0xE2, 0x77, 0x78}); + check_bon8(1000000u, {0xF0, 0x07, 0x33, 0x18}); + check_bon8(-9, {0xC0}); + check_bon8(-100, {0xC3, 0xD9}); + check_bon8(-100000, {0xE5, 0xFF, 0x15}); + check_bon8(-1000000, {0xF0, 0xCB, 0x3A, 0xB5}); + } + + SECTION("every integer from -300000 to 600000") + { + for (std::int64_t i = -300000; i <= 600000; ++i) + { + const json j = i; + const auto packed = json::to_bon8(j); + CHECK(packed.size() == (i >= -10 && i <= 39 ? 1u : i >= -1930 && i <= 3879 ? 2u : i >= -264074 && i <= 528167 ? 3u : 4u)); + const json decoded = json::from_bon8(packed); + if (decoded != j) + { + CAPTURE(i) + CHECK(decoded == j); + } + } + } + + SECTION("signed values are read back as unsigned when not negative") + { + const json j = json::from_bon8(json::to_bon8(json(static_cast(1000)))); + CHECK(j.is_number_unsigned()); + CHECK(j == 1000); + } + + SECTION("unsigned integers above int64") + { + json _; + const json j = 9223372036854775808u; + CHECK_THROWS_WITH_AS(_ = json::to_bon8(j), "[json.exception.out_of_range.407] integer number 9223372036854775808 cannot be represented by BON8 as it does not fit int64", json::out_of_range&); + } + } + + SECTION("floating-point numbers") + { + SECTION("one-byte values") + { + check_bon8(-1.0, {0xFB}); + check_bon8(0.0, {0xFC}); + check_bon8(1.0, {0xFD}); + } + + SECTION("binary32") + { + check_bon8(2.0, {0x8E, 0x40, 0x00, 0x00, 0x00}); + check_bon8(0.5, {0x8E, 0x3F, 0x00, 0x00, 0x00}); + check_bon8(-1.5, {0x8E, 0xBF, 0xC0, 0x00, 0x00}); + check_bon8(3.4028234663852886e38, {0x8E, 0x7F, 0x7F, 0xFF, 0xFF}); + } + + SECTION("binary64") + { + check_bon8(100000000.1, {0x8F, 0x41, 0x97, 0xD7, 0x84, 0x00, 0x66, 0x66, 0x66}); + check_bon8(3.14159, {0x8F, 0x40, 0x09, 0x21, 0xF9, 0xF0, 0x1B, 0x86, 0x6E}); + check_bon8(1e300, {0x8F, 0x7E, 0x37, 0xE4, 0x3C, 0x88, 0x00, 0x75, 0x9C}); + } + + SECTION("-0.0 is written as binary32") + { + const json j = -0.0; + CHECK(json::to_bon8(j) == bytes{0x8E, 0x80, 0x00, 0x00, 0x00}); + const json decoded = json::from_bon8(json::to_bon8(j)); + CHECK(decoded.get() == 0.0); + CHECK(std::signbit(decoded.get())); + } + + SECTION("infinity is written as binary32") + { + check_bon8(std::numeric_limits::infinity(), {0x8E, 0x7F, 0x80, 0x00, 0x00}); + check_bon8(-std::numeric_limits::infinity(), {0x8E, 0xFF, 0x80, 0x00, 0x00}); + } + + SECTION("NaN is written as binary32 0x7F800001") + { + const json j = std::numeric_limits::quiet_NaN(); + CHECK(json::to_bon8(j) == bytes{0x8E, 0x7F, 0x80, 0x00, 0x01}); + CHECK(std::isnan(json::from_bon8(json::to_bon8(j)).get())); + } + } + + SECTION("strings") + { + SECTION("empty string") + { + check_bon8("", {0xFF}); + } + + SECTION("ASCII") + { + check_bon8("a", {'a', 0xFF}); + check_bon8("This is a string.", concat(bytes{'T', 'h', 'i', 's', ' ', 'i', 's', ' ', 'a', ' ', 's', 't', 'r', 'i', 'n', 'g', '.'}, {0xFF})); + check_bon8(str({0x00}), {0x00, 0xFF}); + check_bon8(str({0x7F}), {0x7F, 0xFF}); + } + + SECTION("multi-byte UTF-8") + { + check_bon8("\xC2\xA3", {0xC2, 0xA3, 0xFF}); // U+00A3 + check_bon8("\xEF\xB8\xBB", {0xEF, 0xB8, 0xBB, 0xFF}); // U+FE3B + check_bon8("\xF0\x9F\x80\x84", {0xF0, 0x9F, 0x80, 0x84, 0xFF}); // U+1F004 + check_bon8("\xC2\x80", {0xC2, 0x80, 0xFF}); // U+0080 + check_bon8("\xDF\xBF", {0xDF, 0xBF, 0xFF}); // U+07FF + check_bon8("\xE0\xA0\x80", {0xE0, 0xA0, 0x80, 0xFF}); // U+0800 + check_bon8("\xED\x9F\xBF", {0xED, 0x9F, 0xBF, 0xFF}); // U+D7FF + check_bon8("\xEE\x80\x80", {0xEE, 0x80, 0x80, 0xFF}); // U+E000 + check_bon8("\xF0\x90\x80\x80", {0xF0, 0x90, 0x80, 0x80, 0xFF}); // U+10000 + check_bon8("\xF4\x8F\xBF\xBF", {0xF4, 0x8F, 0xBF, 0xBF, 0xFF}); // U+10FFFF + check_bon8("a\xC2\xA3" "b", {'a', 0xC2, 0xA3, 'b', 0xFF}); + } + + SECTION("invalid UTF-8 cannot be written") + { + json _; + CHECK_THROWS_WITH_AS(_ = json::to_bon8(str({0x80})), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0x80", json::type_error&); + CHECK_THROWS_WITH_AS(_ = json::to_bon8(str({'a', 0xC0, 0x80})), "[json.exception.type_error.316] invalid UTF-8 byte at index 1: 0xC0", json::type_error&); + CHECK_THROWS_WITH_AS(_ = json::to_bon8(str({0xE0, 0x80, 0x80})), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xE0", json::type_error&); + CHECK_THROWS_WITH_AS(_ = json::to_bon8(str({0xED, 0xA0, 0x80})), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xED", json::type_error&); + CHECK_THROWS_WITH_AS(_ = json::to_bon8(str({0xF4, 0x90, 0x80, 0x80})), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xF4", json::type_error&); + CHECK_THROWS_WITH_AS(_ = json::to_bon8(str({0xF5, 0x80, 0x80, 0x80})), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xF5", json::type_error&); + CHECK_THROWS_WITH_AS(_ = json::to_bon8(str({0xC2, 'a'})), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC2", json::type_error&); + CHECK_THROWS_WITH_AS(_ = json::to_bon8(str({'a', 0xE2, 0x82})), "[json.exception.type_error.316] invalid UTF-8 byte at index 1: 0xE2", json::type_error&); + CHECK_THROWS_WITH_AS(_ = json::to_bon8(json::object({{str({0xFF}), 1}})), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&); + // after a run of ASCII characters that is checked 8 bytes at a time + CHECK_THROWS_WITH_AS(_ = json::to_bon8(std::string(17, 'a') + str({0xC0})), "[json.exception.type_error.316] invalid UTF-8 byte at index 17: 0xC0", json::type_error&); + CHECK_THROWS_WITH_AS(_ = json::to_bon8(std::string(8, 'a') + "\xC3\xA4" + std::string(8, 'a') + str({0xE2, 0x82})), "[json.exception.type_error.316] invalid UTF-8 byte at index 18: 0xE2", json::type_error&); + } + } + + SECTION("arrays") + { + check_bon8(json::array(), {0x80}); + check_bon8({false}, {0x81, 0xF8}); + check_bon8({false, nullptr}, {0x82, 0xF8, 0xFA}); + check_bon8({false, nullptr, true}, {0x83, 0xF8, 0xFA, 0xF9}); + check_bon8({false, nullptr, true, 1.0}, {0x84, 0xF8, 0xFA, 0xF9, 0xFD}); + check_bon8({false, nullptr, true, 1.0, json::array(), 0.0}, {0x85, 0xF8, 0xFA, 0xF9, 0xFD, 0x80, 0xFC, 0xFE}); + check_bon8({{{1}}}, {0x81, 0x81, 0x81, 0x91}); + check_bon8({{{"foo"}}}, {0x81, 0x81, 0x81, 'f', 'o', 'o', 0xFF}); + check_bon8({{{""}}}, {0x81, 0x81, 0x81, 0xFF}); + + SECTION("large array") + { + json j = json::array(); + bytes expected = {0x85}; + for (int i = 0; i < 1000; ++i) + { + j.push_back(nullptr); + expected.push_back(0xFA); + } + expected.push_back(0xFE); + check_bon8(j, expected); + } + } + + SECTION("objects") + { + check_bon8(json::object(), {0x86}); + check_bon8({{"foo", nullptr}}, {0x87, 'f', 'o', 'o', 0xFA}); + check_bon8({{"", true}, {"foo", nullptr}}, {0x88, 0xFF, 0xF9, 'f', 'o', 'o', 0xFA}); + check_bon8({{"a", 1}, {"b", 2}, {"c", 3}}, {0x89, 'a', 0x91, 'b', 0x92, 'c', 0x93}); + check_bon8({{"a", 1}, {"b", 2}, {"c", 3}, {"d", 4}}, {0x8A, 'a', 0x91, 'b', 0x92, 'c', 0x93, 'd', 0x94}); + const json five = {{"one", 1}, {"two", 2}, {"three", 3}, {"four", 4}, {"five", 5}}; + check_bon8(five, {0x8B, 'f', 'i', 'v', 'e', 0x95, 'f', 'o', 'u', 'r', 0x94, 'o', 'n', 'e', 0x91, 't', 'h', 'r', 'e', 'e', 0x93, 't', 'w', 'o', 0x92, 0xFE}); + } + + SECTION("binary values are written as arrays of integers") + { + CHECK(json::to_bon8(json::binary({})) == bytes{0x80}); + CHECK(json::to_bon8(json::binary({0x00, 0x27, 0x28, 0xFF})) == bytes{0x84, 0x90, 0xB7, 0xC2, 0x00, 0xC3, 0x57}); + CHECK(json::to_bon8(json::binary({1, 2, 3, 4, 5}, 42)) == bytes{0x85, 0x91, 0x92, 0x93, 0x94, 0x95, 0xFE}); + CHECK(json::to_bon8({"a", json::binary({1})}) == bytes{0x82, 'a', 0x81, 0x91}); + CHECK(json::from_bon8(json::to_bon8(json::binary({1, 2, 3, 4, 5}))) == json({1, 2, 3, 4, 5})); + } + } + + SECTION("examples from the specification") + { + check_bon8("ab", {'a', 'b', 0xFF}); + check_bon8({"ab", "bc"}, {0x82, 'a', 'b', 0xFF, 'b', 'c', 0xFF}); + check_bon8({"a", "b", "c", "d", "e"}, {0x85, 'a', 0xFF, 'b', 0xFF, 'c', 0xFF, 'd', 0xFF, 'e', 0xFE}); + check_bon8({{"ab", 1}, {"bc", 2}}, {0x88, 'a', 'b', 0x91, 'b', 'c', 0x92}); + // the specification prints this example without the end-of-string + // markers after "b" and "c", which its own rules require: without + // them, the bytes read as the single string "bcd" + check_bon8({{"a", {"b", "c"}}, {"d", 1}}, {0x88, 'a', 0x82, 'b', 0xFF, 'c', 0xFF, 'd', 0x91}); + check_bon8({{"", 1}, {"a", 2}}, {0x88, 0xFF, 0x91, 'a', 0x92}); + } + + SECTION("examples from the discussion of the specification (#2980)") + { + check_bon8({{"a", ""}, {"c", "d"}}, {0x88, 'a', 0xFF, 0xFF, 'c', 0xFF, 'd', 0xFF}); + } + + SECTION("end of strings") + { + SECTION("a string ends at the first byte of an integer") + { + check_bon8({{"a", 100}}, {0x87, 'a', 0xC2, 0x3C}); + check_bon8({{"a", -9}}, {0x87, 'a', 0xC0}); + check_bon8({{"a", -10}}, {0x87, 'a', 0xC1}); + check_bon8({{"a", 5}}, {0x87, 'a', 0x95}); + check_bon8({{"a", 100000}}, {0x87, 'a', 0xE2, 0x77, 0x78}); + check_bon8({{"a", -1000000}}, {0x87, 'a', 0xF0, 0xCB, 0x3A, 0xB5}); + check_bon8({{"a", 67637032}}, {0x87, 'a', 0x8C, 0x04, 0x08, 0x0F, 0x28}); + check_bon8({{"a", 2147483648}}, {0x87, 'a', 0x8D, 0x00, 0x00, 0x00, 0x00, 0x80, 0x00, 0x00, 0x00}); + check_bon8({"\xC2\xA3", 100}, {0x82, 0xC2, 0xA3, 0xC2, 0x3C}); + check_bon8({"\xF0\x9F\x80\x84", -1000000}, {0x82, 0xF0, 0x9F, 0x80, 0x84, 0xF0, 0xCB, 0x3A, 0xB5}); + } + + SECTION("a string ends at the first byte of other values") + { + check_bon8({"a", 2.0}, {0x82, 'a', 0x8E, 0x40, 0x00, 0x00, 0x00}); + check_bon8({"a", 0.1}, {0x82, 'a', 0x8F, 0x3F, 0xB9, 0x99, 0x99, 0x99, 0x99, 0x99, 0x9A}); + check_bon8({"a", nullptr, "b", true, "c", false}, {0x85, 'a', 0xFA, 'b', 0xF9, 'c', 0xF8, 0xFE}); + check_bon8({"a", -1.0, "b", 0.0, "c", 1.0}, {0x85, 'a', 0xFB, 'b', 0xFC, 'c', 0xFD, 0xFE}); + check_bon8({"a", json::array()}, {0x82, 'a', 0x80}); + check_bon8({"a", json::object()}, {0x82, 'a', 0x86}); + check_bon8({{"a", {1, 2, 3, 4, 5}}}, {0x87, 'a', 0x85, 0x91, 0x92, 0x93, 0x94, 0x95, 0xFE}); + check_bon8({{"a", {{"b", 1}, {"c", 2}, {"d", 3}, {"e", 4}, {"f", 5}}}}, {0x87, 'a', 0x8B, 'b', 0x91, 'c', 0x92, 'd', 0x93, 'e', 0x94, 'f', 0x95, 0xFE}); + } + + SECTION("a string ends at the end of its container") + { + check_bon8({1, 2, 3, 4, "e"}, {0x85, 0x91, 0x92, 0x93, 0x94, 'e', 0xFE}); + check_bon8({{"a", 1}, {"b", 2}, {"c", 3}, {"d", 4}, {"e", "f"}}, {0x8B, 'a', 0x91, 'b', 0x92, 'c', 0x93, 'd', 0x94, 'e', 0xFF, 'f', 0xFE}); + check_bon8({{1, 2, 3, 4, "e"}, 1}, {0x82, 0x85, 0x91, 0x92, 0x93, 0x94, 'e', 0xFE, 0x91}); + check_bon8({{1, 2, 3, 4, "\xC2\xA3"}, 100}, {0x82, 0x85, 0x91, 0x92, 0x93, 0x94, 0xC2, 0xA3, 0xFE, 0xC2, 0x3C}); + } + + SECTION("a string that another string follows is terminated") + { + check_bon8({"s", "s"}, {0x82, 's', 0xFF, 's', 0xFF}); + check_bon8({"", "s"}, {0x82, 0xFF, 's', 0xFF}); + check_bon8({"s", ""}, {0x82, 's', 0xFF, 0xFF}); + check_bon8({"", ""}, {0x82, 0xFF, 0xFF}); + check_bon8({{"a", "b"}, {"c", "d"}}, {0x88, 'a', 0xFF, 'b', 0xFF, 'c', 0xFF, 'd', 0xFF}); + check_bon8({{"a", ""}}, {0x87, 'a', 0xFF, 0xFF}); + check_bon8({{"", ""}}, {0x87, 0xFF, 0xFF}); + } + + SECTION("a container that ends with a string and that a string follows") + { + check_bon8({{"a"}, "b"}, {0x82, 0x81, 'a', 0xFF, 'b', 0xFF}); + check_bon8({{{{"a"}}}, "b"}, {0x82, 0x81, 0x81, 0x81, 'a', 0xFF, 'b', 0xFF}); + check_bon8({{"a", {"x"}}, {"b", 1}}, {0x88, 'a', 0x81, 'x', 0xFF, 'b', 0x91}); + check_bon8({{"a", {{"b", "c"}}}, {"d", 1}}, {0x88, 'a', 0x87, 'b', 0xFF, 'c', 0xFF, 'd', 0x91}); + check_bon8({{"a"}, {"b"}}, {0x82, 0x81, 'a', 0x81, 'b', 0xFF}); + } + + SECTION("a string that ends the message is terminated") + { + check_bon8({"a"}, {0x81, 'a', 0xFF}); + check_bon8({{"a", "b"}}, {0x87, 'a', 0xFF, 'b', 0xFF}); + check_bon8({{{"a"}}}, {0x81, 0x81, 0x81, 'a', 0xFF}); + } + } + + SECTION("non-canonical input is accepted") + { + // an integer with a longer encoding than necessary + CHECK(json::from_bon8(bytes{0x8C, 0x00, 0x00, 0x00, 0x01}) == 1); + CHECK(json::from_bon8(bytes{0x8D, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF}) == -1); + // a float with a longer encoding than necessary + CHECK(json::from_bon8(bytes{0x8F, 0x3F, 0xF0, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00}) == 1.0); + CHECK(json::from_bon8(bytes{0x8E, 0x00, 0x00, 0x00, 0x00}) == 0.0); + // an unsized container with few elements + CHECK(json::from_bon8(bytes{0x85, 0x91, 0x92, 0xFE}) == json({1, 2})); + CHECK(json::from_bon8(bytes{0x85, 0xFE}) == json::array()); + CHECK(json::from_bon8(bytes{0x8B, 'a', 0x91, 0xFE}) == json({{"a", 1}})); + // unsorted keys + CHECK(json::from_bon8(bytes{0x88, 'b', 0x91, 'a', 0x92}) == json({{"a", 2}, {"b", 1}})); + // an end-of-string marker that is not needed + CHECK(json::from_bon8(bytes{0x82, 'a', 0xFF, 0x91}) == json({"a", 1})); + CHECK(json::from_bon8(bytes{0x87, 'a', 0xFF, 0x91}) == json({{"a", 1}})); + } + + SECTION("types of values read") + { + CHECK(json::from_bon8(bytes{0x90}).is_number_unsigned()); + CHECK(json::from_bon8(bytes{0xC2, 0x00}).is_number_unsigned()); + CHECK(json::from_bon8(bytes{0x8C, 0x00, 0x00, 0x00, 0x00}).is_number_unsigned()); + CHECK(json::from_bon8(bytes{0xB8}).is_number_integer()); + CHECK(!json::from_bon8(bytes{0xB8}).is_number_unsigned()); + CHECK(json::from_bon8(bytes{0xC2, 0xC0}).is_number_integer()); + CHECK(json::from_bon8(bytes{0xFC}).is_number_float()); + CHECK(json::from_bon8(bytes{0x8E, 0x40, 0x00, 0x00, 0x00}).is_number_float()); + CHECK(json::from_bon8(bytes{0xFF}).is_string()); + } + + SECTION("errors") + { + json _; + + SECTION("empty input") + { + CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes()), "[json.exception.parse_error.110] parse error at byte 1: syntax error while parsing BON8 value: unexpected end of input", json::parse_error&); + CHECK(json::from_bon8(bytes(), true, false).is_discarded()); + } + + SECTION("too short numbers") + { + CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0x8C}), "[json.exception.parse_error.110] parse error at byte 2: syntax error while parsing BON8 number: unexpected end of input", json::parse_error&); + CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0x8C, 0x00, 0x00, 0x00}), "[json.exception.parse_error.110] parse error at byte 5: syntax error while parsing BON8 number: unexpected end of input", json::parse_error&); + CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0x8D, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00}), "[json.exception.parse_error.110] parse error at byte 9: syntax error while parsing BON8 number: unexpected end of input", json::parse_error&); + CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0x8E, 0x00}), "[json.exception.parse_error.110] parse error at byte 3: syntax error while parsing BON8 number: unexpected end of input", json::parse_error&); + CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0x8F, 0x00}), "[json.exception.parse_error.110] parse error at byte 3: syntax error while parsing BON8 number: unexpected end of input", json::parse_error&); + CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0xC2}), "[json.exception.parse_error.110] parse error at byte 2: syntax error while parsing BON8 number: unexpected end of input", json::parse_error&); + CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0xE0, 0x00}), "[json.exception.parse_error.110] parse error at byte 3: syntax error while parsing BON8 number: unexpected end of input", json::parse_error&); + CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0xF0, 0x00, 0x00}), "[json.exception.parse_error.110] parse error at byte 4: syntax error while parsing BON8 number: unexpected end of input", json::parse_error&); + CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0x82, 'a', 0xF0, 0xC0}), "[json.exception.parse_error.110] parse error at byte 5: syntax error while parsing BON8 number: unexpected end of input", json::parse_error&); + } + + SECTION("unterminated strings") + { + CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{'a'}), "[json.exception.parse_error.110] parse error at byte 2: syntax error while parsing BON8 string: unexpected end of input", json::parse_error&); + CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0x81, 'a', 'b'}), "[json.exception.parse_error.110] parse error at byte 4: syntax error while parsing BON8 string: unexpected end of input", json::parse_error&); + CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0xF0, 0x9F, 0x80, 0x84}), "[json.exception.parse_error.110] parse error at byte 5: syntax error while parsing BON8 string: unexpected end of input", json::parse_error&); + CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0xF0, 0x9F, 0x80}), "[json.exception.parse_error.110] parse error at byte 4: syntax error while parsing BON8 string: unexpected end of input", json::parse_error&); + CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0x87, 'a'}), "[json.exception.parse_error.110] parse error at byte 3: syntax error while parsing BON8 string: unexpected end of input", json::parse_error&); + } + + SECTION("invalid UTF-8") + { + // overlong + CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0xE0, 0x80, 0x80, 0xFF}), "[json.exception.parse_error.112] parse error at byte 2: syntax error while parsing BON8 string: invalid UTF-8 byte: 0x80", json::parse_error&); + CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0xF0, 0x80, 0x80, 0x80, 0xFF}), "[json.exception.parse_error.112] parse error at byte 2: syntax error while parsing BON8 string: invalid UTF-8 byte: 0x80", json::parse_error&); + // surrogate + CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0xED, 0xA0, 0x80, 0xFF}), "[json.exception.parse_error.112] parse error at byte 2: syntax error while parsing BON8 string: invalid UTF-8 byte: 0xA0", json::parse_error&); + // above U+10FFFF + CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0xF4, 0x90, 0x80, 0x80, 0xFF}), "[json.exception.parse_error.112] parse error at byte 2: syntax error while parsing BON8 string: invalid UTF-8 byte: 0x90", json::parse_error&); + CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0xF5, 0x80, 0x80, 0x80, 0xFF}), "[json.exception.parse_error.112] parse error at byte 2: syntax error while parsing BON8 string: invalid UTF-8 byte: 0x80", json::parse_error&); + // missing continuation byte + CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0xE2, 0x82, 'a', 0xFF}), "[json.exception.parse_error.112] parse error at byte 3: syntax error while parsing BON8 string: invalid UTF-8 byte: 0x61", json::parse_error&); + CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0x81, 'a', 0xE2, 0x82, 'b'}), "[json.exception.parse_error.112] parse error at byte 5: syntax error while parsing BON8 string: invalid UTF-8 byte: 0x62", json::parse_error&); + } + + SECTION("end-of-container marker where a value is expected") + { + CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0xFE}), "[json.exception.parse_error.112] parse error at byte 1: syntax error while parsing BON8 value: invalid byte: 0xFE", json::parse_error&); + CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0x81, 0xFE}), "[json.exception.parse_error.112] parse error at byte 2: syntax error while parsing BON8 value: invalid byte: 0xFE", json::parse_error&); + CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0x8B, 'a', 0xFE}), "[json.exception.parse_error.112] parse error at byte 3: syntax error while parsing BON8 value: invalid byte: 0xFE", json::parse_error&); + } + + SECTION("keys that are not strings") + { + CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0x87, 0x91, 0x91}), "[json.exception.parse_error.112] parse error at byte 2: syntax error while parsing BON8 key: expected a string; last byte: 0x91", json::parse_error&); + CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0x87, 0xFA, 0x91}), "[json.exception.parse_error.112] parse error at byte 2: syntax error while parsing BON8 key: expected a string; last byte: 0xFA", json::parse_error&); + CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0x87, 0x80, 0x91}), "[json.exception.parse_error.112] parse error at byte 2: syntax error while parsing BON8 key: expected a string; last byte: 0x80", json::parse_error&); + CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0x87, 0xFE}), "[json.exception.parse_error.112] parse error at byte 2: syntax error while parsing BON8 key: expected a string; last byte: 0xFE", json::parse_error&); + // an integer that begins with a UTF-8 lead byte + CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0x87, 0xC2, 0x00, 0x91}), "[json.exception.parse_error.112] parse error at byte 2: syntax error while parsing BON8 key: expected a string; last byte: 0xC2", json::parse_error&); + CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0x88, 'a', 0x91, 0xF0, 0xC0, 0x00, 0x00, 0x91}), "[json.exception.parse_error.112] parse error at byte 4: syntax error while parsing BON8 key: expected a string; last byte: 0xF0", json::parse_error&); + CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0x87}), "[json.exception.parse_error.110] parse error at byte 2: syntax error while parsing BON8 key: unexpected end of input", json::parse_error&); + } + + SECTION("unterminated containers") + { + CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0x82, 0x91}), "[json.exception.parse_error.110] parse error at byte 3: syntax error while parsing BON8 value: unexpected end of input", json::parse_error&); + CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0x85, 0x91}), "[json.exception.parse_error.110] parse error at byte 3: syntax error while parsing BON8 value: unexpected end of input", json::parse_error&); + CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0x8B, 'a', 0x91}), "[json.exception.parse_error.110] parse error at byte 4: syntax error while parsing BON8 key: unexpected end of input", json::parse_error&); + CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0x87, 'a'}), "[json.exception.parse_error.110] parse error at byte 3: syntax error while parsing BON8 string: unexpected end of input", json::parse_error&); + CHECK_THROWS_WITH_AS(_ = json::from_bon8(bytes{0x87, 'a', 0xFF}), "[json.exception.parse_error.110] parse error at byte 4: syntax error while parsing BON8 value: unexpected end of input", json::parse_error&); + } + + SECTION("strict mode") + { + const bytes vec = {0x90, 0x90}; + CHECK(json::from_bon8(vec, false) == 0); + CHECK_THROWS_WITH_AS(_ = json::from_bon8(vec), "[json.exception.parse_error.110] parse error at byte 2: syntax error while parsing BON8 value: expected end of input; last byte: 0x90", json::parse_error&); + CHECK(json::from_bon8(vec, true, false).is_discarded()); + + // the byte after a string that ends a container + const bytes vec2 = {0x81, 'a', 0x90}; + CHECK(json::from_bon8(vec2, false) == json({"a"})); + CHECK_THROWS_WITH_AS(_ = json::from_bon8(vec2), "[json.exception.parse_error.110] parse error at byte 3: syntax error while parsing BON8 value: expected end of input; last byte: 0x90", json::parse_error&); + + const bytes vec3 = {0x81, 'a', 0xC2, 0x00}; + CHECK(json::from_bon8(vec3, false) == json({"a"})); + CHECK_THROWS_WITH_AS(_ = json::from_bon8(vec3), "[json.exception.parse_error.110] parse error at byte 3: syntax error while parsing BON8 value: expected end of input; last byte: 0xC2", json::parse_error&); + } + } + + SECTION("SAX aborts") + { + SECTION("start_array(len)") + { + const bytes v = {0x83, 0x91, 0x92, 0x93}; + SaxCountdown scp(0); + CHECK(!json::sax_parse(v, &scp, json::input_format_t::bon8)); + } + + SECTION("start_array()") + { + const bytes v = {0x85, 0x91, 0xFE}; + SaxCountdown scp(0); + CHECK(!json::sax_parse(v, &scp, json::input_format_t::bon8)); + } + + SECTION("end_array()") + { + const bytes v = {0x85, 0x91, 0xFE}; + SaxCountdown scp(2); + CHECK(!json::sax_parse(v, &scp, json::input_format_t::bon8)); + } + + SECTION("start_object(len)") + { + const bytes v = {0x87, 'f', 'o', 'o', 0xF8}; + SaxCountdown scp(0); + CHECK(!json::sax_parse(v, &scp, json::input_format_t::bon8)); + } + + SECTION("key()") + { + const bytes v = {0x87, 'f', 'o', 'o', 0xF8}; + SaxCountdown scp(1); + CHECK(!json::sax_parse(v, &scp, json::input_format_t::bon8)); + } + + SECTION("end_object()") + { + const bytes v = {0x87, 'f', 'o', 'o', 0xF8}; + SaxCountdown scp(3); + CHECK(!json::sax_parse(v, &scp, json::input_format_t::bon8)); + } + + SECTION("values") + { + const std::vector values = + { + {0xFA}, {0xF9}, {0x91}, {0xB8}, {0xC2, 0x00}, {0xC2, 0xC0}, {0xE0, 0x00, 0x00}, + {0x8C, 0, 0, 0, 1}, {0x8D, 0, 0, 0, 0, 0, 0, 0, 1}, {0xFB}, {0xFC}, {0xFD}, + {0x8E, 0x40, 0, 0, 0}, {0x8F, 0x40, 0, 0, 0, 0, 0, 0, 0}, {'a', 0xFF}, {0xFF} + }; + for (const auto& v : values) + { + SaxCountdown scp(0); + CHECK(!json::sax_parse(v, &scp, json::input_format_t::bon8)); + } + } + } +} + +// the test catches the exceptions of invalid input +#if !defined(JSON_NOEXCEPTION) +TEST_CASE("BON8 strings from contiguous and stream input") +{ + // contiguous input copies the valid UTF-8 of a string in bulk, a stream + // is read byte by byte; both must end strings and report errors alike + const std::string ascii(20, 'a'); + const std::vector inputs = + { + // the string ends at 0xFF, at a marker, and at an integer + concat(concat(bytes(ascii.begin(), ascii.end()), {0xC3, 0xA4}), {0xFF}), + concat(concat({0x82}, bytes(ascii.begin(), ascii.end())), {0x91}), + concat(concat({0x85}, bytes(ascii.begin(), ascii.end())), {0xFE}), + concat(concat({0x82}, bytes(ascii.begin(), ascii.end())), {0xC2, 0x05}), + concat(concat({0x87}, bytes(ascii.begin(), ascii.end())), {0xE2, 0x82, 0xAC, 0xF0, 0x05, 0x00, 0x00}), + // invalid UTF-8 and a premature end after a run of valid characters + concat(bytes(ascii.begin(), ascii.end()), {0xE0, 0x80, 0x80}), + concat(bytes(ascii.begin(), ascii.end()), {0xE2, 0x82, 0x2F}), + concat(bytes(ascii.begin(), ascii.end()), {0xE2, 0x82}), + bytes(ascii.begin(), ascii.end()), + // trailing bytes after a string that ends a container + concat(concat({0x81}, bytes(ascii.begin(), ascii.end())), {0x91}), + }; + + for (const auto& input : inputs) + { + CAPTURE(input) + std::string from_vector; + std::string from_stream; + try + { + from_vector = json::from_bon8(input).dump(); + } + catch (const json::parse_error& e) + { + from_vector = e.what(); + } + try + { + std::istringstream stream(str(input)); + from_stream = json::from_bon8(stream).dump(); + } + catch (const json::parse_error& e) + { + from_stream = e.what(); + } + CHECK(from_vector == from_stream); + } + + json _; + CHECK(json::from_bon8(inputs[0]) == ascii + "\xC3\xA4"); + CHECK(json::from_bon8(inputs[3]) == json({ascii, 45})); + CHECK_THROWS_WITH_AS(_ = json::from_bon8(inputs[5]), "[json.exception.parse_error.112] parse error at byte 22: syntax error while parsing BON8 string: invalid UTF-8 byte: 0x80", json::parse_error&); + CHECK_THROWS_WITH_AS(_ = json::from_bon8(inputs[6]), "[json.exception.parse_error.112] parse error at byte 23: syntax error while parsing BON8 string: invalid UTF-8 byte: 0x2F", json::parse_error&); + CHECK_THROWS_WITH_AS(_ = json::from_bon8(inputs[7]), "[json.exception.parse_error.110] parse error at byte 23: syntax error while parsing BON8 string: unexpected end of input", json::parse_error&); + CHECK_THROWS_WITH_AS(_ = json::from_bon8(inputs[9]), "[json.exception.parse_error.110] parse error at byte 22: syntax error while parsing BON8 value: expected end of input; last byte: 0x91", json::parse_error&); +} +#endif + +// use this testcase outside [hide] to run it with Valgrind +TEST_CASE("BON8 nesting does not consume the call stack") +{ + // Note that deeply nested values must not be compared, copied or dumped + // here: those operations are still recursive. Depth is measured by + // descending instead. + + SECTION("an unterminated chain is reported, not crashed on") + { + json _; + const bytes input(300000, 0x81); + CHECK_THROWS_WITH_AS(_ = json::from_bon8(input), "[json.exception.parse_error.110] parse error at byte 300001: syntax error while parsing BON8 value: unexpected end of input", json::parse_error&); + CHECK(json::from_bon8(input, true, false).is_discarded()); + } + + SECTION("a well-formed deep value is read through the SAX interface") + { + bytes input(300000, 0x85); + input.push_back(0x91); // innermost value + input.insert(input.end(), 300000, 0xFE); + + SaxCountdown accept_all(600001); + CHECK(json::sax_parse(input, &accept_all, json::input_format_t::bon8)); + } + + SECTION("a well-formed deep value is read into a value") + { + const std::size_t depth = 10000; + bytes input; + for (std::size_t i = 0; i < depth; ++i) + { + input.push_back(0x87); + input.push_back('a'); + } + // the innermost key is followed by a string, so it is terminated + input.push_back(0xFF); + input.push_back('b'); + input.push_back(0xFF); + + json j = json::from_bon8(input); + + std::size_t measured = 0; + const json* p = &j; + while (p->is_object() && !p->empty()) + { + p = &p->at("a"); + ++measured; + } + CHECK(measured == depth); + CHECK(*p == "b"); + } +} + +TEST_CASE("single BON8 roundtrip") +{ + SECTION("sample.json") + { + std::string const filename = TEST_DATA_DIRECTORY "/json_testsuite/sample.json"; + + // parse JSON file + std::ifstream f_json(filename); + const json j1 = json::parse(f_json); + + // parse BON8 file + auto packed = utils::read_binary_file(filename + ".bon8"); + json j2; + CHECK_NOTHROW(j2 = json::from_bon8(packed)); + + // compare parsed JSON values + CHECK(j1 == j2); + + SECTION("roundtrips") + { + SECTION("std::ostringstream") + { + std::basic_ostringstream ss; + json::to_bon8(j1, ss); + json j3 = json::from_bon8(ss.str()); + CHECK(j1 == j3); + } + + SECTION("std::string") + { + std::string s; + json::to_bon8(j1, s); + json j3 = json::from_bon8(s); + CHECK(j1 == j3); + } + } + + // check with different start index + packed.insert(packed.begin(), 5, 0xff); + CHECK(j1 == json::from_bon8(packed.begin() + 5, packed.end())); + } +} + +TEST_CASE("Parse BON8 directly from a file using iterator and sentinel") +{ + std::string const filename = TEST_DATA_DIRECTORY "/json_testsuite/sample.json.bon8"; + std::ifstream file(filename, std::ios::binary); + const std::istreambuf_iterator first(file); + const json parsed = json::from_bon8(first, utils::istreambuf_sentinel{}); + CHECK((parsed.is_object() || parsed.is_array())); +} + +TEST_CASE("BON8 roundtrips" * doctest::skip()) +{ + SECTION("input from HikoGUI") + { + // The .bon8 files were created with the BON8 encoder of HikoGUI + // (https://github.com/hikoworks/hikogui), the reference implementation + // of the format. Both implementations sort object keys, so the output + // is compared byte for byte. + for (std::string filename : + { + TEST_DATA_DIRECTORY "/json_nlohmann_tests/all_unicode.json", + TEST_DATA_DIRECTORY "/json.org/1.json", + TEST_DATA_DIRECTORY "/json.org/2.json", + TEST_DATA_DIRECTORY "/json.org/3.json", + TEST_DATA_DIRECTORY "/json.org/4.json", + TEST_DATA_DIRECTORY "/json.org/5.json", + TEST_DATA_DIRECTORY "/json_roundtrip/roundtrip01.json", + TEST_DATA_DIRECTORY "/json_roundtrip/roundtrip02.json", + TEST_DATA_DIRECTORY "/json_roundtrip/roundtrip03.json", + TEST_DATA_DIRECTORY "/json_roundtrip/roundtrip04.json", + TEST_DATA_DIRECTORY "/json_roundtrip/roundtrip05.json", + TEST_DATA_DIRECTORY "/json_roundtrip/roundtrip06.json", + TEST_DATA_DIRECTORY "/json_roundtrip/roundtrip07.json", + TEST_DATA_DIRECTORY "/json_roundtrip/roundtrip08.json", + TEST_DATA_DIRECTORY "/json_roundtrip/roundtrip09.json", + TEST_DATA_DIRECTORY "/json_roundtrip/roundtrip10.json", + TEST_DATA_DIRECTORY "/json_roundtrip/roundtrip11.json", + TEST_DATA_DIRECTORY "/json_roundtrip/roundtrip12.json", + TEST_DATA_DIRECTORY "/json_roundtrip/roundtrip13.json", + TEST_DATA_DIRECTORY "/json_roundtrip/roundtrip14.json", + TEST_DATA_DIRECTORY "/json_roundtrip/roundtrip15.json", + TEST_DATA_DIRECTORY "/json_roundtrip/roundtrip16.json", + TEST_DATA_DIRECTORY "/json_roundtrip/roundtrip17.json", + TEST_DATA_DIRECTORY "/json_roundtrip/roundtrip18.json", + TEST_DATA_DIRECTORY "/json_roundtrip/roundtrip19.json", + TEST_DATA_DIRECTORY "/json_roundtrip/roundtrip20.json", + TEST_DATA_DIRECTORY "/json_roundtrip/roundtrip21.json", + TEST_DATA_DIRECTORY "/json_roundtrip/roundtrip22.json", + TEST_DATA_DIRECTORY "/json_roundtrip/roundtrip23.json", + TEST_DATA_DIRECTORY "/json_roundtrip/roundtrip24.json", + TEST_DATA_DIRECTORY "/json_roundtrip/roundtrip25.json", + TEST_DATA_DIRECTORY "/json_roundtrip/roundtrip26.json", + TEST_DATA_DIRECTORY "/json_roundtrip/roundtrip27.json", + TEST_DATA_DIRECTORY "/json_roundtrip/roundtrip28.json", + TEST_DATA_DIRECTORY "/json_roundtrip/roundtrip29.json", + TEST_DATA_DIRECTORY "/json_roundtrip/roundtrip30.json", + TEST_DATA_DIRECTORY "/json_roundtrip/roundtrip31.json", + TEST_DATA_DIRECTORY "/json_roundtrip/roundtrip32.json", + TEST_DATA_DIRECTORY "/json_testsuite/sample.json", + TEST_DATA_DIRECTORY "/json_tests/pass1.json", + TEST_DATA_DIRECTORY "/json_tests/pass2.json", + TEST_DATA_DIRECTORY "/json_tests/pass3.json", + TEST_DATA_DIRECTORY "/regression/floats.json", + TEST_DATA_DIRECTORY "/regression/signed_ints.json", + TEST_DATA_DIRECTORY "/regression/working_file.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_array_arraysWithSpaces.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_array_empty-string.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_array_empty.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_array_ending_with_newline.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_array_false.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_array_heterogeneous.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_array_null.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_array_with_1_and_newline.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_array_with_leading_space.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_array_with_several_null.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_array_with_trailing_space.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_number.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_number_0e+1.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_number_0e1.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_number_after_space.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_number_double_close_to_zero.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_number_double_huge_neg_exp.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_number_int_with_exp.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_number_minus_zero.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_number_negative_int.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_number_negative_one.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_number_negative_zero.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_number_real_capital_e.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_number_real_capital_e_neg_exp.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_number_real_capital_e_pos_exp.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_number_real_exponent.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_number_real_fraction_exponent.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_number_real_neg_exp.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_number_real_pos_exponent.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_number_real_underflow.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_number_simple_int.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_number_simple_real.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_number_too_big_neg_int.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_number_too_big_pos_int.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_number_very_big_negative_int.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_object.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_object_basic.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_object_duplicated_key.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_object_duplicated_key_and_value.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_object_empty.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_object_empty_key.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_object_escaped_null_in_key.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_object_extreme_numbers.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_object_long_strings.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_object_simple.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_object_string_unicode.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_object_with_newlines.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_string_1_2_3_bytes_UTF-8_sequences.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_string_UTF-16_Surrogates_U+1D11E_MUSICAL_SYMBOL_G_CLEF.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_string_accepted_surrogate_pair.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_string_accepted_surrogate_pairs.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_string_allowed_escapes.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_string_backslash_and_u_escaped_zero.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_string_backslash_doublequotes.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_string_comments.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_string_double_escape_a.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_string_double_escape_n.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_string_escaped_control_character.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_string_escaped_noncharacter.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_string_in_array.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_string_in_array_with_leading_space.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_string_last_surrogates_1_and_2.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_string_newline_uescaped.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_string_nonCharacterInUTF-8_U+10FFFF.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_string_nonCharacterInUTF-8_U+1FFFF.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_string_nonCharacterInUTF-8_U+FFFF.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_string_null_escape.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_string_one-byte-utf-8.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_string_pi.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_string_simple_ascii.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_string_space.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_string_three-byte-utf-8.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_string_two-byte-utf-8.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_string_u+2028_line_sep.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_string_u+2029_par_sep.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_string_uEscape.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_string_unescaped_char_delete.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_string_unicode.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_string_unicodeEscapedBackslash.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_string_unicode_2.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_string_unicode_U+200B_ZERO_WIDTH_SPACE.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_string_unicode_U+2064_invisible_plus.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_string_unicode_escaped_double_quote.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_string_utf8.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_string_with_del_character.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_structure_lonely_false.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_structure_lonely_int.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_structure_lonely_negative_real.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_structure_lonely_null.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_structure_lonely_string.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_structure_lonely_true.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_structure_string_empty.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_structure_trailing_newline.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_structure_true_in_array.json", + TEST_DATA_DIRECTORY "/nst_json_testsuite/test_parsing/y_structure_whitespace_array.json" + }) + { + CAPTURE(filename) + + // parse JSON file + std::ifstream f_json(filename); + const json j1 = json::parse(f_json); + + // parse BON8 file + const auto packed = utils::read_binary_file(filename + ".bon8"); + + { + INFO_WITH_TEMP(filename + ": std::vector"); + json j2; + CHECK_NOTHROW(j2 = json::from_bon8(packed)); + CHECK(j1 == j2); + } + + { + INFO_WITH_TEMP(filename + ": std::ifstream"); + std::ifstream f_bon8(filename + ".bon8", std::ios::binary); + json j2; + CHECK_NOTHROW(j2 = json::from_bon8(f_bon8)); + CHECK(j1 == j2); + } + + { + INFO_WITH_TEMP(filename + ": iterator pair"); + json j2; + CHECK_NOTHROW(j2 = json::from_bon8(packed.begin(), packed.end())); + CHECK(j1 == j2); + } + + { + INFO_WITH_TEMP(filename + ": output adapters: std::vector"); + std::vector vec; + json::to_bon8(j1, vec); + CHECK(vec == packed); + } + } + } +} + +#ifdef JSON_HAS_CPP_17 +TEST_CASE("BON8 with std::byte") +{ + SECTION("vector roundtrip") + { + const json original = + { + {"name", "test"}, + {"value", 42}, + {"array", {1, 2, 3}} + }; + + const std::vector temp = json::to_bon8(original); + std::vector bon8_data(temp.size()); + for (size_t i = 0; i < temp.size(); ++i) + { + bon8_data[i] = std::byte(temp[i]); + } + + json from_bytes; + CHECK_NOTHROW(from_bytes = json::from_bon8(bon8_data)); + CHECK(from_bytes == original); + } + + SECTION("empty vector") + { + const std::vector empty_data; + CHECK_THROWS_WITH_AS([&]() + { + [[maybe_unused]] auto result = json::from_bon8(empty_data); + return true; + } + (), + "[json.exception.parse_error.110] parse error at byte 1: syntax error while parsing BON8 value: unexpected end of input", + json::parse_error&); + } +} +#endif diff --git a/tests/src/unit-bson.cpp b/tests/src/unit-bson.cpp index 81331b43a..292ff4dc1 100644 --- a/tests/src/unit-bson.cpp +++ b/tests/src/unit-bson.cpp @@ -1066,6 +1066,45 @@ TEST_CASE("Incomplete BSON Input") } } +// the test catches the exceptions of invalid input +#if !defined(JSON_NOEXCEPTION) +TEST_CASE("BSON keys from contiguous and stream input") +{ + // contiguous input reads a key up to its \x00-byte in one step, a stream + // reads it byte by byte; both must give the same value or error for the + // complete document and for every truncation of it + const json j = {{"", true}, {"k", {1, 2, 3}}, {std::string(40, 'x'), {{"nested key", "value"}}}}; + const std::vector bson = json::to_bson(j); + CHECK(json::from_bson(bson) == j); + + for (std::size_t length = 0; length <= bson.size(); ++length) + { + CAPTURE(length) + const std::vector input(bson.begin(), bson.begin() + static_cast(length)); + std::string from_vector; + std::string from_stream; + try + { + from_vector = json::from_bson(input).dump(); + } + catch (const json::parse_error& e) + { + from_vector = e.what(); + } + try + { + std::istringstream stream(std::string(input.begin(), input.end())); + from_stream = json::from_bson(stream).dump(); + } + catch (const json::parse_error& e) + { + from_stream = e.what(); + } + CHECK(from_vector == from_stream); + } +} +#endif + TEST_CASE("Negative size of binary value") { // invalid BSON: the size of the binary value is -1 diff --git a/tests/src/unit-class_parser.cpp b/tests/src/unit-class_parser.cpp index 707f23c18..3d825162e 100644 --- a/tests/src/unit-class_parser.cpp +++ b/tests/src/unit-class_parser.cpp @@ -2761,7 +2761,7 @@ TEST_CASE("diagnostic positions: value lifetime, input adapters, and SAX") SECTION("binary formats have no text positions") { - // binary formats (CBOR, MessagePack, UBJSON, BSON, BJData) are + // binary formats (BJData, BON8, BSON, CBOR, MessagePack, UBJSON) are // parsed via detail::binary_reader, which never sets // start_position/end_position on the values it produces (they // have no notion of a text offset), so every value's position @@ -2778,6 +2778,10 @@ TEST_CASE("diagnostic positions: value lifetime, input adapters, and SAX") CHECK(from_msgpack.start_pos() == std::string::npos); CHECK(from_msgpack.end_pos() == std::string::npos); + const json from_bon8 = json::from_bon8(json::to_bon8(src)); + CHECK(from_bon8.start_pos() == std::string::npos); + CHECK(from_bon8.end_pos() == std::string::npos); + const json from_ubjson = json::from_ubjson(json::to_ubjson(src)); CHECK(from_ubjson.start_pos() == std::string::npos); CHECK(from_ubjson.end_pos() == std::string::npos); diff --git a/tests/src/unit-custom-binary-type.cpp b/tests/src/unit-custom-binary-type.cpp index efedba3cd..d885376cb 100644 --- a/tests/src/unit-custom-binary-type.cpp +++ b/tests/src/unit-custom-binary-type.cpp @@ -84,6 +84,8 @@ TEST_CASE("binary type whose value type is not std::uint8_t") // UBJSON has no binary type, so binary values are written as an array CHECK(byte_binary_json::from_ubjson(byte_binary_json::to_ubjson(j)) == byte_binary_json({0, 1, 255})); + // the same holds for BON8 + CHECK(byte_binary_json::from_bon8(byte_binary_json::to_bon8(j)) == byte_binary_json({0, 1, 255})); } #endif } diff --git a/tests/src/unit-custom-object-type.cpp b/tests/src/unit-custom-object-type.cpp index cb2cb5ff3..a579e0b2d 100644 --- a/tests/src/unit-custom-object-type.cpp +++ b/tests/src/unit-custom-object-type.cpp @@ -297,6 +297,7 @@ TEST_CASE("object type without key_compare") const auto j = no_key_compare_json::parse(R"({"a":[1,2,3],"b":"x"})"); CHECK(no_key_compare_json::from_cbor(no_key_compare_json::to_cbor(j)) == j); CHECK(no_key_compare_json::from_msgpack(no_key_compare_json::to_msgpack(j)) == j); + CHECK(no_key_compare_json::from_bon8(no_key_compare_json::to_bon8(j)) == j); } SECTION("flatten and unflatten") diff --git a/tests/src/unit-explicit_instantiation.cpp b/tests/src/unit-explicit_instantiation.cpp new file mode 100644 index 000000000..71132a59f --- /dev/null +++ b/tests/src/unit-explicit_instantiation.cpp @@ -0,0 +1,37 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +// cmake/test.cmake selects the C++ standard versions with which to build a +// unit test based on the presence of JSON_HAS_CPP_ macros. +// The regression below only showed on C++17, so build this file for every +// standard like the other regression tests: +// JSON_HAS_CPP_17 JSON_HAS_CPP_20 (do not remove; see note at top of file) + +#include "doctest_compatibility.h" + +#include +using json = nlohmann::json; + +///////////////////////////////////////////////////////////////////// +// for #4825 - explicitly instantiating basic_json must compile; this +// forces instantiation of binary_writer::write_bjdata_ndarray, whose +// static_cast was ambiguous under explicit instantiation on +// C++17. Merely compiling this translation unit is the regression test. +// +// The instantiation compiles every member function, so it has a file of its +// own: in unit-regression3.cpp it made the object too large for the MinGW +// linker to relocate (see #5511). +///////////////////////////////////////////////////////////////////// +template class nlohmann::basic_json<>; + +TEST_CASE("explicit instantiation of basic_json (#4825)") +{ + const json j = {1, "two", 3.0}; + CHECK(j.size() == 3); + CHECK(json::from_bjdata(json::to_bjdata(j)) == j); +} diff --git a/tests/src/unit-ordered_json2.cpp b/tests/src/unit-ordered_json2.cpp index 83eb7668d..653820628 100644 --- a/tests/src/unit-ordered_json2.cpp +++ b/tests/src/unit-ordered_json2.cpp @@ -326,6 +326,15 @@ TEST_CASE("ordered_json across binary formats") CHECK(collect_keys(restored) == original_keys); CHECK(collect_keys(restored["mango"]) == original_mango_keys); } + + SECTION("BON8") + { + const auto bytes = ordered_json::to_bon8(original); + const auto restored = ordered_json::from_bon8(bytes); + CHECK(restored == original); + CHECK(collect_keys(restored) == original_keys); + CHECK(collect_keys(restored["mango"]) == original_mango_keys); + } } TEST_CASE("alt_json (custom string_t) across binary formats") @@ -353,6 +362,13 @@ TEST_CASE("alt_json (custom string_t) across binary formats") CHECK(restored == original); } + SECTION("BON8") + { + const auto bytes = alt_json::to_bon8(original); + const auto restored = alt_json::from_bon8(bytes); + CHECK(restored == original); + } + SECTION("BSON") { const auto bytes = alt_json::to_bson(original); diff --git a/tests/src/unit-regression2.cpp b/tests/src/unit-regression2.cpp index b128b7a73..c466958bf 100644 --- a/tests/src/unit-regression2.cpp +++ b/tests/src/unit-regression2.cpp @@ -332,6 +332,7 @@ TEST_CASE("regression tests 2") CHECK(float_json::from_cbor(float_json::to_cbor(j)) == j); CHECK(float_json::from_msgpack(float_json::to_msgpack(j)) == j); CHECK(float_json::from_ubjson(float_json::to_ubjson(j)) == j); + CHECK(float_json::from_bon8(float_json::to_bon8(j)) == j); float_json j2 = {1000.0, 2000.0, 3000.0}; CHECK(float_json::from_ubjson(float_json::to_ubjson(j2, true, true)) == j2); diff --git a/tests/src/unit-regression3.cpp b/tests/src/unit-regression3.cpp index 882b3866b..1c2ebd3f0 100644 --- a/tests/src/unit-regression3.cpp +++ b/tests/src/unit-regression3.cpp @@ -74,13 +74,7 @@ using ordered_json = nlohmann::ordered_json; #endif #endif -///////////////////////////////////////////////////////////////////// -// for #4825 - explicitly instantiating basic_json must compile; this -// forces instantiation of binary_writer::write_bjdata_ndarray, whose -// static_cast was ambiguous under explicit instantiation on -// C++17. Merely compiling this translation unit is the regression test. -///////////////////////////////////////////////////////////////////// -template class nlohmann::basic_json<>; +// the explicit instantiation for #4825 is in unit-explicit_instantiation.cpp ///////////////////////////////////////////////////////////////////// // for #4440 @@ -894,6 +888,7 @@ TEST_CASE("regression test #5476 - array type without reserve()") // the binary formats pass a definite length to start_array() CHECK(deque_json::from_cbor(deque_json::to_cbor(j)) == j); CHECK(deque_json::from_msgpack(deque_json::to_msgpack(j)) == j); + CHECK(deque_json::from_bon8(deque_json::to_bon8(j)) == j); // parse() instantiates the callback parser as well, which reserves too const auto with_callback = deque_json::parse(R"([1,2,3])", [](int /*depth*/, deque_json::parse_event_t /*event*/, deque_json& /*parsed*/) noexcept From 509c07041f4f35daa52f5cc48c9041eda69b84bf Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Sun, 27 Sep 2026 17:50:19 +0200 Subject: [PATCH 64/64] Fix CI: clang-tidy and clang/libstdc++ 10 in the MessagePack size tests (#5599) The tests added by #5515 fail two ways on develop: - clang-tidy reports the size() overrides of huge_string and huge_binary (readability-convert-member-functions-to-static) and the non-const test value (misc-const-correctness); mark them like the #5584 types - clang with libstdc++ 10 cannot compile the file for C++17: the std::filesystem::path conversion considered for huge_string, a class derived from std::string, is ambiguous. Guard it with JSON_TEST_BEYOND_UINT32_STRING, which #5584 introduced for the same reason, and define that macro before both test blocks. Signed-off-by: Niels Lohmann --- tests/src/unit-msgpack.cpp | 23 ++++++++++++----------- 1 file changed, 12 insertions(+), 11 deletions(-) diff --git a/tests/src/unit-msgpack.cpp b/tests/src/unit-msgpack.cpp index e7616b27f..077073052 100644 --- a/tests/src/unit-msgpack.cpp +++ b/tests/src/unit-msgpack.cpp @@ -2165,6 +2165,13 @@ TEST_CASE("MessagePack with std::byte") #endif // the fake sizes below do not fit into a 32-bit std::size_t +// with clang and libstdc++ 10, the std::filesystem::path conversion that +// C++17 builds consider for every string type is ambiguous for a class +// derived from std::string, so the string case is not tested there +#if !(defined(__clang__) && defined(_GLIBCXX_RELEASE) && _GLIBCXX_RELEASE < 11) + #define JSON_TEST_BEYOND_UINT32_STRING 1 +#endif + #if SIZE_MAX > UINT32_MAX template> struct huge_array : std::vector @@ -2262,11 +2269,12 @@ TEST_CASE("MessagePack Size above uint32 for object") object.fake_size = false; } +#ifdef JSON_TEST_BEYOND_UINT32_STRING struct huge_string : std::string { using std::string::string; - std::size_t size() const noexcept + std::size_t size() const noexcept // NOLINT(readability-convert-member-functions-to-static) { return static_cast(UINT32_MAX) + 1ULL; } @@ -2287,20 +2295,20 @@ using huge_string_json = nlohmann::basic_json < TEST_CASE("MessagePack Size above uint32 for string") { - - huge_string_json j = "hello"; + const huge_string_json j = "hello"; CHECK_THROWS_WITH_AS( huge_string_json::to_msgpack(j), "[json.exception.out_of_range.412] MessagePack length 4294967296 exceeds maximum of 4294967295", json::out_of_range&); } +#endif struct huge_binary : std::vector { using std::vector::vector; - std::size_t size() const noexcept + std::size_t size() const noexcept // NOLINT(readability-convert-member-functions-to-static) { return static_cast(UINT32_MAX) + 1ULL; } @@ -2355,13 +2363,6 @@ class beyond_uint32_binary_t : public std::vector } }; -// with clang and libstdc++ 10, the std::filesystem::path conversion that -// C++17 builds consider for every string type is ambiguous for a class -// derived from std::string, so the string case is not tested there -#if !(defined(__clang__) && defined(_GLIBCXX_RELEASE) && _GLIBCXX_RELEASE < 11) - #define JSON_TEST_BEYOND_UINT32_STRING 1 -#endif - #ifdef JSON_TEST_BEYOND_UINT32_STRING class beyond_uint32_string_t : public std::string {